From 7e96edbcf0ae34ffe9bd9dceecb3b58680228d3c Mon Sep 17 00:00:00 2001 From: Morten Hjorth-Jensen Date: Thu, 23 Sep 2021 08:34:01 +0200 Subject: [PATCH] update --- doc/pub/week38/html/._week38-bs000.html | 159 +++++----- doc/pub/week38/html/._week38-bs001.html | 159 +++++----- doc/pub/week38/html/._week38-bs002.html | 159 +++++----- doc/pub/week38/html/._week38-bs003.html | 159 +++++----- doc/pub/week38/html/._week38-bs004.html | 159 +++++----- doc/pub/week38/html/._week38-bs005.html | 159 +++++----- doc/pub/week38/html/._week38-bs006.html | 159 +++++----- doc/pub/week38/html/._week38-bs007.html | 159 +++++----- doc/pub/week38/html/._week38-bs008.html | 159 +++++----- doc/pub/week38/html/._week38-bs009.html | 159 +++++----- doc/pub/week38/html/._week38-bs010.html | 159 +++++----- doc/pub/week38/html/._week38-bs011.html | 159 +++++----- doc/pub/week38/html/._week38-bs012.html | 159 +++++----- doc/pub/week38/html/._week38-bs013.html | 161 +++++----- doc/pub/week38/html/._week38-bs014.html | 187 ++++++----- doc/pub/week38/html/._week38-bs015.html | 231 +++++++------- doc/pub/week38/html/._week38-bs016.html | 236 +++++++------- doc/pub/week38/html/._week38-bs017.html | 281 +++++++---------- doc/pub/week38/html/._week38-bs018.html | 226 +++++++------- doc/pub/week38/html/._week38-bs019.html | 215 +++++++------ doc/pub/week38/html/._week38-bs020.html | 252 ++++++++------- doc/pub/week38/html/._week38-bs021.html | 312 +++++++++++-------- doc/pub/week38/html/._week38-bs022.html | 288 ++++++----------- doc/pub/week38/html/._week38-bs023.html | 188 ++++++----- doc/pub/week38/html/._week38-bs024.html | 223 ++++++------- doc/pub/week38/html/._week38-bs025.html | 231 +++++++------- doc/pub/week38/html/._week38-bs026.html | 219 ++++++------- doc/pub/week38/html/._week38-bs027.html | 188 +++++------ doc/pub/week38/html/._week38-bs028.html | 189 ++++++----- doc/pub/week38/html/._week38-bs029.html | 193 ++++++------ doc/pub/week38/html/._week38-bs030.html | 192 ++++++------ doc/pub/week38/html/._week38-bs031.html | 189 ++++++----- doc/pub/week38/html/._week38-bs032.html | 231 ++++++++------ doc/pub/week38/html/._week38-bs033.html | 242 +++++++------- doc/pub/week38/html/._week38-bs034.html | 203 ++++++------ doc/pub/week38/html/._week38-bs035.html | 234 ++++++++------ doc/pub/week38/html/._week38-bs036.html | 231 ++++++-------- doc/pub/week38/html/._week38-bs037.html | 180 ++++++----- doc/pub/week38/html/._week38-bs038.html | 183 ++++++----- doc/pub/week38/html/._week38-bs039.html | 180 ++++++----- doc/pub/week38/html/._week38-bs040.html | 182 ++++++----- doc/pub/week38/html/._week38-bs041.html | 176 +++++------ doc/pub/week38/html/._week38-bs042.html | 181 +++++------ doc/pub/week38/html/week38-bs.html | 159 +++++----- doc/pub/week38/html/week38-reveal.html | 77 ++--- doc/pub/week38/html/week38-solarized.html | 80 ++--- doc/pub/week38/html/week38.html | 80 ++--- doc/pub/week38/ipynb/ipynb-week38-src.tar.gz | Bin 192 -> 193 bytes doc/pub/week38/ipynb/week38.ipynb | 84 ++--- doc/src/week38/week38.do.txt | 76 ++--- 50 files changed, 4303 insertions(+), 4744 deletions(-) diff --git a/doc/pub/week38/html/._week38-bs000.html b/doc/pub/week38/html/._week38-bs000.html index 406c60fa3..d5042016a 100644 --- a/doc/pub/week38/html/._week38-bs000.html +++ b/doc/pub/week38/html/._week38-bs000.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -466,7 +461,7 @@ MathJax.Hub.Config({
  • 9
  • 10
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs001.html b/doc/pub/week38/html/._week38-bs001.html index 8267f0ab5..af8f75e3e 100644 --- a/doc/pub/week38/html/._week38-bs001.html +++ b/doc/pub/week38/html/._week38-bs001.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -446,7 +441,7 @@ MathJax.Hub.Config({
  • 10
  • 11
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs002.html b/doc/pub/week38/html/._week38-bs002.html index bbcd424db..91c52aa41 100644 --- a/doc/pub/week38/html/._week38-bs002.html +++ b/doc/pub/week38/html/._week38-bs002.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -443,7 +438,7 @@ MathJax.Hub.Config({
  • 11
  • 12
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs003.html b/doc/pub/week38/html/._week38-bs003.html index 95d9b4914..9d45adb15 100644 --- a/doc/pub/week38/html/._week38-bs003.html +++ b/doc/pub/week38/html/._week38-bs003.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -497,7 +492,7 @@ $$
  • 12
  • 13
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs004.html b/doc/pub/week38/html/._week38-bs004.html index 48fda37a8..7c57ad552 100644 --- a/doc/pub/week38/html/._week38-bs004.html +++ b/doc/pub/week38/html/._week38-bs004.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -461,7 +456,7 @@ cross-validation (LOOCV).
  • 13
  • 14
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs005.html b/doc/pub/week38/html/._week38-bs005.html index 3546d505f..e7722ffe2 100644 --- a/doc/pub/week38/html/._week38-bs005.html +++ b/doc/pub/week38/html/._week38-bs005.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -473,7 +468,7 @@ $$
  • 14
  • 15
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs006.html b/doc/pub/week38/html/._week38-bs006.html index 61671ce61..7efb0d77b 100644 --- a/doc/pub/week38/html/._week38-bs006.html +++ b/doc/pub/week38/html/._week38-bs006.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -464,7 +459,7 @@ For the various values of \( k \)
  • 15
  • 16
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs007.html b/doc/pub/week38/html/._week38-bs007.html index a3b304fc9..15b115048 100644 --- a/doc/pub/week38/html/._week38-bs007.html +++ b/doc/pub/week38/html/._week38-bs007.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -543,7 +538,7 @@ plt.show()
  • 16
  • 17
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs008.html b/doc/pub/week38/html/._week38-bs008.html index 2e94ceb3a..750e13d02 100644 --- a/doc/pub/week38/html/._week38-bs008.html +++ b/doc/pub/week38/html/._week38-bs008.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -470,7 +465,7 @@ from the library Scikit-Learn (when not shrinking \( \beta_0 \)) for the
  • 17
  • 18
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs009.html b/doc/pub/week38/html/._week38-bs009.html index 582a61dd7..5bc92cd2b 100644 --- a/doc/pub/week38/html/._week38-bs009.html +++ b/doc/pub/week38/html/._week38-bs009.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -479,7 +474,7 @@ This can clearly lead to problems in evaluating the cost/loss functions.
  • 18
  • 19
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs010.html b/doc/pub/week38/html/._week38-bs010.html index 814cd2d6d..7183ff19a 100644 --- a/doc/pub/week38/html/._week38-bs010.html +++ b/doc/pub/week38/html/._week38-bs010.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -473,7 +468,7 @@ y_pred = y_pred 19
  • 20
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs011.html b/doc/pub/week38/html/._week38-bs011.html index 1e85555ce..a2a36a0d1 100644 --- a/doc/pub/week38/html/._week38-bs011.html +++ b/doc/pub/week38/html/._week38-bs011.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -485,7 +480,7 @@ $$
  • 20
  • 21
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs012.html b/doc/pub/week38/html/._week38-bs012.html index 21cab17d8..fd50ab60c 100644 --- a/doc/pub/week38/html/._week38-bs012.html +++ b/doc/pub/week38/html/._week38-bs012.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -511,7 +506,7 @@ $$
  • 21
  • 22
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs013.html b/doc/pub/week38/html/._week38-bs013.html index 5eee7cab0..981ac928b 100644 --- a/doc/pub/week38/html/._week38-bs013.html +++ b/doc/pub/week38/html/._week38-bs013.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -443,7 +438,7 @@ $$ $$

    -What does this mean for thus? And why do we insist on all this? Let us look at some examples. +What does this mean? And why do we insist on all this? Let us look at some examples.

    @@ -471,7 +466,7 @@ What does this mean for thus? And why do we insist on all this? Let us look at s

  • 22
  • 23
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs014.html b/doc/pub/week38/html/._week38-bs014.html index 63e354f36..3deb087c8 100644 --- a/doc/pub/week38/html/._week38-bs014.html +++ b/doc/pub/week38/html/._week38-bs014.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -439,6 +434,10 @@ Note also that we do not split the data into training and test. np.random.seed(2021) +def MSE(y_data,y_model): + n = np.size(y_model) + return np.sum((y_data-y_model)**2)/n + def fit_beta(X, y): return np.linalg.pinv(X.T @ X) @ X.T @ y @@ -466,6 +465,12 @@ clf = LinearRegression(fit_interceptprint(f"True beta: {true_beta}") print(f"Fitted beta: {beta}") print(f"Sklearn fitted beta: {clf.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with intercept column: {intercept}") +print(MSE(y,ypredictOwn) +print(f"MSE with intercept column from SKL: {intercept}") +print(MSE(y,ypredictSKL) plt.figure() @@ -494,6 +499,12 @@ intercept = np. print(f"Fitted beta (wiothout intercept): {beta}") print(f"Sklearn intercept: {clf.intercept_}") print(f"Sklearn fitted beta (without intercept): {clf.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with Manual intercept: {intercept}") +print(MSE(y,ypredictOwn) +print(f"MSE with Sklearn intercept: {clf.intercept_}") +print(MSE(y,ypredictSKL) plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)") plt.plot(x, clf.predict(X), "--", label="Sklearn (fit_intercept=True)") @@ -502,6 +513,18 @@ plt.legend() plt.show() +

    +The intercept is the value of our output/target variable +when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). + +

    +Printing the MSE, we see first that both methods give the same +MSE. However, changing the value of the intercept gives a larger or +smaller MSE, meaning that the MSE is penalized by the value of the +intercept! Setting the intercept to true in the Scikit-Learn +function or simply scaling our results, leads to a fit which is +independent of the specific value of the intercept. +

    @@ -528,7 +551,7 @@ plt.show()

  • 23
  • 24
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs015.html b/doc/pub/week38/html/._week38-bs015.html index 3e263491e..ac68c1ea3 100644 --- a/doc/pub/week38/html/._week38-bs015.html +++ b/doc/pub/week38/html/._week38-bs015.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,13 +418,10 @@ MathJax.Hub.Config({ -

    Another straight Line, the Intercept and Scaling again

    +

    Code Examples

    -As we said above, the intercept is the value of our output/target variable -when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). -Let us now study a case where we again fit a straight line including the intercept in the design matrix -using ordinary least squares. We are only interested in the MSE value here. +Armed with this wisdom, we attempt first simply set the intercept eqault to False in our implementation of Ridge regression for yet another vanilla data set.

    @@ -444,10 +436,6 @@ using ordinary least squares. We are only interested in the MSE value here. n = np.size(y_model) return np.sum((y_data-y_model)**2)/n -def fit_beta(X, y): - return np.linalg.pinv(X.T @ X) @ X.T @ y - - # A seed just to ensure that the random numbers are the same for every run. # Useful for eventual debugging. @@ -455,36 +443,55 @@ np.random.= 100 x = np.random.rand(n) -y = 10.0+5*x +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) - -Maxpolydegree = 2 +Maxpolydegree = 20 X = np.zeros((n,Maxpolydegree)) - +#We include explicitely the intercept column for degree in range(Maxpolydegree): X[:,degree] = x**degree - - # We split the data in test and training data X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2) -# Intercept is included in the design matrix -# own code first -OwnBeta = fit_beta(X_train, y_train) -# Scikit-Learn -skl = LinearRegression(fit_intercept=False).fit(X_train, y_train) -ypredictOwn = X_test @ OwnBeta -ypredictSKL = skl.predict(X_test) -print(MSE(y_test,ypredictOwn) -print(MSE(y_test,ypredictSKL) +p = Maxpolydegree +I = np.eye(p,p) +# Decide which values of lambda to use +nlambdas = 4 +MSEOwnRidgePredict = np.zeros(nlambdas) +MSERidgePredict = np.zeros(nlambdas) +lambdas = np.logspace(-4, 4, nlambdas) +for i in range(nlambdas): + lmb = lambdas[i] + OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train + # include lasso using Scikit-Learn + # Note: we include the intercept column and no scaling + RegRidge = linear_model.Ridge(lmb,fit_intercept=False) + RegRidge.fit(X_train,y_train) + # and then make the prediction + ytildeOwnRidge = X_train @ OwnRidgeBeta + ypredictOwnRidge = X_test @ OwnRidgeBeta + ytildeRidge = RegRidge.predict(X_train) + ypredictRidge = RegRidge.predict(X_test) + MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge) + MSERidgePredict[i] = MSE(y_test,ypredictRidge) + print("Beta values for own Ridge implementation") + print(OwnRidgeBeta) + print("Beta values for Scikit-Learn Ridge implementation") + print(RegRidge.coef_) +# Now plot the results +plt.figure() +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test') +plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test') + +plt.xlabel('log10(lambda)') +plt.ylabel('MSE') +plt.legend() +plt.show()

    -Printing the MSE, we see first that both methods give the same -MSE. However, changing the value of the intercept gives a larger or -smaller MSE, meaning that the MSE is penalized by the value of the -intercept! Setting the intercept to true in the Scikit-Learn -function or simply scaling our results, leads to a fit which is -independent of the specific value of the intercept. +The results here agree when we force Scikit-Learn's Ridge function to include the first column in our design matrix. +We see that the results agree very well. What happens if we do not include the intercept in our fit? +Let us see how we can change this code by zero centering.

    @@ -512,7 +519,7 @@ independent of the specific value of the intercept.

  • 24
  • 25
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs016.html b/doc/pub/week38/html/._week38-bs016.html index 88ee61c24..f6be53d27 100644 --- a/doc/pub/week38/html/._week38-bs016.html +++ b/doc/pub/week38/html/._week38-bs016.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,11 +418,7 @@ MathJax.Hub.Config({ -

    Code Examples

    - -

    -Armed with this wisdom, we attempt first simply set the intercept eqault to False in our implementation of Ridge regression for yet another vanilla data set. - +

    Taking out the mean

    @@ -436,68 +427,87 @@ Armed with this wisdom, we attempt first simply set the intercept eqault to F import matplotlib.pyplot as plt from sklearn.model_selection import train_test_split from sklearn import linear_model +from sklearn.preprocessing import StandardScaler def MSE(y_data,y_model): n = np.size(y_model) return np.sum((y_data-y_model)**2)/n - - # A seed just to ensure that the random numbers are the same for every run. # Useful for eventual debugging. -np.random.seed(3155) +np.random.seed(315) n = 100 x = np.random.rand(n) y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) Maxpolydegree = 20 -X = np.zeros((n,Maxpolydegree)) -#We include explicitely the intercept column -for degree in range(Maxpolydegree): - X[:,degree] = x**degree +X = np.zeros((n,Maxpolydegree-1)) + +for degree in range(1,Maxpolydegree): #No intercept column + X[:,degree-1] = x**(degree) + # We split the data in test and training data X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2) -p = Maxpolydegree + + + + +#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable +X_train_mean = np.mean(X_train,axis=0) +#Center by removing mean from each feature +X_train_scaled = X_train - X_train_mean +X_test_scaled = X_test - X_train_mean +#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered) +#Remove the intercept from the training data. +y_scaler = np.mean(y_train) +y_train_scaled = y_train - y_scaler + + +p = Maxpolydegree-1 I = np.eye(p,p) # Decide which values of lambda to use nlambdas = 4 MSEOwnRidgePredict = np.zeros(nlambdas) MSERidgePredict = np.zeros(nlambdas) -lambdas = np.logspace(-4, 4, nlambdas) + +lambdas = np.logspace(-4, 1, nlambdas) for i in range(nlambdas): lmb = lambdas[i] - OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train - # include lasso using Scikit-Learn - # Note: we include the intercept column and no scaling - RegRidge = linear_model.Ridge(lmb,fit_intercept=False) + OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled) + intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data + #Add intercept to prediction + ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ + #EQUIVALENT PREDICTION: + #Add intercept to prediction + ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler + print("Values for own Ridge prediction") + print(ypredictOwnRidge) + RegRidge = linear_model.Ridge(lmb) RegRidge.fit(X_train,y_train) - # and then make the prediction - ytildeOwnRidge = X_train @ OwnRidgeBeta - ypredictOwnRidge = X_test @ OwnRidgeBeta - ytildeRidge = RegRidge.predict(X_train) ypredictRidge = RegRidge.predict(X_test) + print("Values for SL Ridge prediction") + print(ypredictRidge) MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge) MSERidgePredict[i] = MSE(y_test,ypredictRidge) print("Beta values for own Ridge implementation") - print(OwnRidgeBeta) + print(OwnRidgeBeta) #Intercept is given by mean of target variable print("Beta values for Scikit-Learn Ridge implementation") print(RegRidge.coef_) + print('Intercept from own implementation:') + print(intercept_) + print('Intercept from Scikit-Learn Ridge implementation') + print(RegRidge.intercept_) + # Now plot the results plt.figure() -plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test') -plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test') - +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test') +plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test') plt.xlabel('log10(lambda)') plt.ylabel('MSE') plt.legend() plt.show() -

    -The results here agree when we force Scikit-Learn's Ridge function to include the first column in our design matrix. -We see that the results agree very well. What happens if we do not include the intercept in our fit? -Let us see how we can change this code by zero centering. -

    @@ -524,7 +534,7 @@ Let us see how we can change this code by zero centering.

  • 25
  • 26
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs017.html b/doc/pub/week38/html/._week38-bs017.html index dc97a19c0..81818dadc 100644 --- a/doc/pub/week38/html/._week38-bs017.html +++ b/doc/pub/week38/html/._week38-bs017.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,96 +418,60 @@ MathJax.Hub.Config({ -

    Taking out the mean

    +

    More complicated Example: The Ising model

    + +

    +The one-dimensional Ising model with nearest neighbor interaction, no +external field and a constant coupling constant \( J \) is given by + +$$ +\begin{align} + H = -J \sum_{k}^L s_k s_{k + 1}, +\tag{1} +\end{align} +$$ + +

    +where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins +in the system is determined by \( L \). For the one-dimensional system +there is no phase transition. + +

    +We will look at a system of \( L = 40 \) spins with a coupling constant of +\( J = 1 \). To get enough training data we will generate 10000 states +with their respective energies. +

    import numpy as np
    -import pandas as pd
     import matplotlib.pyplot as plt
    +from mpl_toolkits.axes_grid1 import make_axes_locatable
    +import seaborn as sns
    +import scipy.linalg as scl
     from sklearn.model_selection import train_test_split
    -from sklearn import linear_model
    -from sklearn.preprocessing import StandardScaler
    +import tqdm
    +sns.set(color_codes=True)
    +cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
     
    -def MSE(y_data,y_model):
    -    n = np.size(y_model)
    -    return np.sum((y_data-y_model)**2)/n
    -# A seed just to ensure that the random numbers are the same for every run.
    -# Useful for eventual debugging.
    -np.random.seed(315)
    +L = 40
    +n = int(1e4)
     
    -n = 100
    -x = np.random.rand(n)
    -y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
    +spins = np.random.choice([-1, 1], size=(n, L))
    +J = 1.0
     
    -Maxpolydegree = 20
    -X = np.zeros((n,Maxpolydegree-1))
    +energies = np.zeros(n)
     
    -for degree in range(1,Maxpolydegree): #No intercept column
    -    X[:,degree-1] = x**(degree)
    -
    -# We split the data in test and training data
    -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    -
    -
    -
    -
    -
    -#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
    -X_train_mean = np.mean(X_train,axis=0)
    -#Center by removing mean from each feature
    -X_train_scaled = X_train - X_train_mean 
    -X_test_scaled = X_test - X_train_mean
    -#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)
    -#Remove the intercept from the training data.
    -y_scaler = np.mean(y_train)           
    -y_train_scaled = y_train - y_scaler   
    -
    -
    -p = Maxpolydegree-1
    -I = np.eye(p,p)
    -# Decide which values of lambda to use
    -nlambdas = 4
    -MSEOwnRidgePredict = np.zeros(nlambdas)
    -MSERidgePredict = np.zeros(nlambdas)
    -
    -lambdas = np.logspace(-4, 1, nlambdas)
    -for i in range(nlambdas):
    -    lmb = lambdas[i]
    -    OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
    -    intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
    -    #Add intercept to prediction
    -    ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ 
    -    #EQUIVALENT PREDICTION:
    -    #Add intercept to prediction
    -    ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler 
    -    print("Values for own Ridge prediction")
    -    print(ypredictOwnRidge)
    -    RegRidge = linear_model.Ridge(lmb)
    -    RegRidge.fit(X_train,y_train)
    -    ypredictRidge = RegRidge.predict(X_test)
    -    print("Values for SL Ridge prediction")
    -    print(ypredictRidge)
    -    MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
    -    MSERidgePredict[i] = MSE(y_test,ypredictRidge)
    -    print("Beta values for own Ridge implementation")
    -    print(OwnRidgeBeta) #Intercept is given by mean of target variable
    -    print("Beta values for Scikit-Learn Ridge implementation")
    -    print(RegRidge.coef_)
    -    print('Intercept from own implementation:')
    -    print(intercept_)
    -    print('Intercept from Scikit-Learn Ridge implementation')
    -    print(RegRidge.intercept_)
    -
    -# Now plot the results
    -plt.figure()
    -plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
    -plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
    -plt.xlabel('log10(lambda)')
    -plt.ylabel('MSE')
    -plt.legend()
    -plt.show()
    +for i in range(n):
    +    energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
     
    +

    +Here we use ordinary least squares +regression to predict the energy for the nearest neighbor +one-dimensional Ising model on a ring, i.e., the endpoints wrap +around. We will use linear regression to fit a value for +the coupling constant to achieve this. +

    @@ -539,7 +498,7 @@ plt.show()

  • 26
  • 27
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs018.html b/doc/pub/week38/html/._week38-bs018.html index 4c735a419..f0f4a3def 100644 --- a/doc/pub/week38/html/._week38-bs018.html +++ b/doc/pub/week38/html/._week38-bs018.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,60 +418,53 @@ MathJax.Hub.Config({ -

    More complicated Example: The Ising model

    +

    Reformulating the problem to suit regression

    -The one-dimensional Ising model with nearest neighbor interaction, no -external field and a constant coupling constant \( J \) is given by +A more general form for the one-dimensional Ising model is $$ \begin{align} - H = -J \sum_{k}^L s_k s_{k + 1}, -\tag{1} + H = - \sum_j^L \sum_k^L s_j s_k J_{jk}. +\tag{2} \end{align} $$

    -where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins -in the system is determined by \( L \). For the one-dimensional system -there is no phase transition. +Here we allow for interactions beyond the nearest neighbors and a state dependent +coupling constant. This latter expression can be formulated as +a matrix-product +$$ +\begin{align} + \boldsymbol{H} = \boldsymbol{X} J, +\tag{3} +\end{align} +$$

    -We will look at a system of \( L = 40 \) spins with a coupling constant of -\( J = 1 \). To get enough training data we will generate 10000 states -with their respective energies. +where \( X_{jk} = s_j s_k \) and \( J \) is a matrix which consists of the +elements \( -J_{jk} \). This form of writing the energy fits perfectly +with the form utilized in linear regression, that is + +$$ +\begin{align} + \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}, +\tag{4} +\end{align} +$$ + +

    +We split the data in training and test data as discussed in the previous example

    -

    import numpy as np
    -import matplotlib.pyplot as plt
    -from mpl_toolkits.axes_grid1 import make_axes_locatable
    -import seaborn as sns
    -import scipy.linalg as scl
    -from sklearn.model_selection import train_test_split
    -import tqdm
    -sns.set(color_codes=True)
    -cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
    -
    -L = 40
    -n = int(1e4)
    -
    -spins = np.random.choice([-1, 1], size=(n, L))
    -J = 1.0
    -
    -energies = np.zeros(n)
    -
    +
    X = np.zeros((n, L ** 2))
     for i in range(n):
    -    energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
    +    X[i] = np.outer(spins[i], spins[i]).ravel()
    +y = energies
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
     
    -

    -Here we use ordinary least squares -regression to predict the energy for the nearest neighbor -one-dimensional Ising model on a ring, i.e., the endpoints wrap -around. We will use linear regression to fit a value for -the coupling constant to achieve this. -

    @@ -503,7 +491,7 @@ the coupling constant to achieve this.

  • 27
  • 28
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs019.html b/doc/pub/week38/html/._week38-bs019.html index 83603bda7..29fa795a5 100644 --- a/doc/pub/week38/html/._week38-bs019.html +++ b/doc/pub/week38/html/._week38-bs019.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,52 +418,50 @@ MathJax.Hub.Config({ -

    Reformulating the problem to suit regression

    +

    Linear regression

    -A more general form for the one-dimensional Ising model is +In the ordinary least squares method we choose the cost function $$ \begin{align} - H = - \sum_j^L \sum_k^L s_j s_k J_{jk}. -\tag{2} + C(\boldsymbol{X}, \boldsymbol{\beta})= \frac{1}{n}\left\{(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})\right\}. +\tag{5} \end{align} $$

    -Here we allow for interactions beyond the nearest neighbors and a state dependent -coupling constant. This latter expression can be formulated as -a matrix-product +We then find the extremal point of \( C \) by taking the derivative with respect to \( \boldsymbol{\beta} \) as discussed above. +This yields the expression for \( \boldsymbol{\beta} \) to be + $$ -\begin{align} - \boldsymbol{H} = \boldsymbol{X} J, -\tag{3} -\end{align} + \boldsymbol{\beta} = \frac{\boldsymbol{X}^T \boldsymbol{y}}{\boldsymbol{X}^T \boldsymbol{X}}, $$

    -where \( X_{jk} = s_j s_k \) and \( J \) is a matrix which consists of the -elements \( -J_{jk} \). This form of writing the energy fits perfectly -with the form utilized in linear regression, that is - -$$ -\begin{align} - \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}, -\tag{4} -\end{align} -$$ - -

    -We split the data in training and test data as discussed in the previous example +which immediately imposes some requirements on \( \boldsymbol{X} \) as there must exist +an inverse of \( \boldsymbol{X}^T \boldsymbol{X} \). If the expression we are modeling contains an +intercept, i.e., a constant term, we must make sure that the +first column of \( \boldsymbol{X} \) consists of \( 1 \). We do this here

    -

    X = np.zeros((n, L ** 2))
    -for i in range(n):
    -    X[i] = np.outer(spins[i], spins[i]).ravel()
    -y = energies
    -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    +
    X_train_own = np.concatenate(
    +    (np.ones(len(X_train))[:, np.newaxis], X_train),
    +    axis=1
    +)
    +X_test_own = np.concatenate(
    +    (np.ones(len(X_test))[:, np.newaxis], X_test),
    +    axis=1
    +)
    +
    +

    + + +

    def ols_inv(x: np.ndarray, y: np.ndarray) -> np.ndarray:
    +    return scl.inv(x.T @ x) @ (x.T @ y)
    +beta = ols_inv(X_train_own, y_train)
     

    @@ -496,7 +489,7 @@ X_train, X_test, y_train, y_test = train_tes

  • 28
  • 29
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs020.html b/doc/pub/week38/html/._week38-bs020.html index 9695ffc89..8c4fc9923 100644 --- a/doc/pub/week38/html/._week38-bs020.html +++ b/doc/pub/week38/html/._week38-bs020.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,51 +418,90 @@ MathJax.Hub.Config({ -

    Linear regression

    +

    Singular Value decomposition

    -In the ordinary least squares method we choose the cost function +Doing the inversion directly turns out to be a bad idea since the matrix +\( \boldsymbol{X}^T\boldsymbol{X} \) is singular. An alternative approach is to use the singular +value decomposition. Using the definition of the Moore-Penrose +pseudoinverse we can write the equation for \( \boldsymbol{\beta} \) as +$$ + \boldsymbol{\beta} = \boldsymbol{X}^{+}\boldsymbol{y}, +$$ + +

    +where the pseudoinverse of \( \boldsymbol{X} \) is given by + +$$ + \boldsymbol{X}^{+} = \frac{\boldsymbol{X}^T}{\boldsymbol{X}^T\boldsymbol{X}}. +$$ + +

    +Using singular value decomposition we can decompose the matrix \( \boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma} \boldsymbol{V}^T \), +where \( \boldsymbol{U} \) and \( \boldsymbol{V} \) are orthogonal(unitary) matrices and \( \boldsymbol{\Sigma} \) contains the singular values (more details below). +where \( X^{+} = V\Sigma^{+} U^T \). This reduces the equation for +\( \omega \) to $$ \begin{align} - C(\boldsymbol{X}, \boldsymbol{\beta})= \frac{1}{n}\left\{(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})\right\}. -\tag{5} + \boldsymbol{\beta} = \boldsymbol{V}\boldsymbol{\Sigma}^{+} \boldsymbol{U}^T \boldsymbol{y}. +\tag{6} \end{align} $$

    -We then find the extremal point of \( C \) by taking the derivative with respect to \( \boldsymbol{\beta} \) as discussed above. -This yields the expression for \( \boldsymbol{\beta} \) to be - -$$ - \boldsymbol{\beta} = \frac{\boldsymbol{X}^T \boldsymbol{y}}{\boldsymbol{X}^T \boldsymbol{X}}, -$$ - -

    -which immediately imposes some requirements on \( \boldsymbol{X} \) as there must exist -an inverse of \( \boldsymbol{X}^T \boldsymbol{X} \). If the expression we are modeling contains an -intercept, i.e., a constant term, we must make sure that the -first column of \( \boldsymbol{X} \) consists of \( 1 \). We do this here +Note that solving this equation by actually doing the pseudoinverse +(which is what we will do) is not a good idea as this operation scales +as \( \mathcal{O}(n^3) \), where \( n \) is the number of elements in a +general matrix. Instead, doing \( QR \)-factorization and solving the +linear system as an equation would reduce this down to +\( \mathcal{O}(n^2) \) operations.

    -

    X_train_own = np.concatenate(
    -    (np.ones(len(X_train))[:, np.newaxis], X_train),
    -    axis=1
    -)
    -X_test_own = np.concatenate(
    -    (np.ones(len(X_test))[:, np.newaxis], X_test),
    -    axis=1
    -)
    +
    def ols_svd(x: np.ndarray, y: np.ndarray) -> np.ndarray:
    +    u, s, v = scl.svd(x)
    +    return v.T @ scl.pinv(scl.diagsvd(s, u.shape[0], v.shape[0])) @ u.T @ y
     

    -

    def ols_inv(x: np.ndarray, y: np.ndarray) -> np.ndarray:
    -    return scl.inv(x.T @ x) @ (x.T @ y)
    -beta = ols_inv(X_train_own, y_train)
    +
    beta = ols_svd(X_train_own,y_train)
     
    +

    +When extracting the \( J \)-matrix we need to make sure that we remove the intercept, as is done here + +

    + + +

    J = beta[1:].reshape(L, L)
    +
    +

    +A way of looking at the coefficients in \( J \) is to plot the matrices as images. + +

    + + +

    fig = plt.figure(figsize=(20, 14))
    +im = plt.imshow(J, **cmap_args)
    +plt.title("OLS", fontsize=18)
    +plt.xticks(fontsize=18)
    +plt.yticks(fontsize=18)
    +cb = fig.colorbar(im)
    +cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
    +plt.show()
    +
    +

    +It is interesting to note that OLS +considers both \( J_{j, j + 1} = -0.5 \) and \( J_{j, j - 1} = -0.5 \) as +valid matrix elements for \( J \). +In our discussion below on hyperparameters and Ridge and Lasso regression we will see that +this problem can be removed, partly and only with Lasso regression. + +

    +In this case our matrix inversion was actually possible. The obvious question now is what is the mathematics behind the SVD? +

    @@ -494,7 +528,7 @@ beta = ols_inv(X_train_own, y_train)

  • 29
  • 30
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs021.html b/doc/pub/week38/html/._week38-bs021.html index 28a951dc3..10aaddbc0 100644 --- a/doc/pub/week38/html/._week38-bs021.html +++ b/doc/pub/week38/html/._week38-bs021.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,74 +418,128 @@ MathJax.Hub.Config({ -

    Singular Value decomposition

    +

    The one-dimensional Ising model

    -Doing the inversion directly turns out to be a bad idea since the matrix -\( \boldsymbol{X}^T\boldsymbol{X} \) is singular. An alternative approach is to use the singular -value decomposition. Using the definition of the Moore-Penrose -pseudoinverse we can write the equation for \( \boldsymbol{\beta} \) as +Let us bring back the Ising model again, but now with an additional +focus on Ridge and Lasso regression as well. We repeat some of the +basic parts of the Ising model and the setup of the training and test +data. The one-dimensional Ising model with nearest neighbor +interaction, no external field and a constant coupling constant \( J \) is +given by -$$ - \boldsymbol{\beta} = \boldsymbol{X}^{+}\boldsymbol{y}, -$$ - -

    -where the pseudoinverse of \( \boldsymbol{X} \) is given by - -$$ - \boldsymbol{X}^{+} = \frac{\boldsymbol{X}^T}{\boldsymbol{X}^T\boldsymbol{X}}. -$$ - -

    -Using singular value decomposition we can decompose the matrix \( \boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma} \boldsymbol{V}^T \), -where \( \boldsymbol{U} \) and \( \boldsymbol{V} \) are orthogonal(unitary) matrices and \( \boldsymbol{\Sigma} \) contains the singular values (more details below). -where \( X^{+} = V\Sigma^{+} U^T \). This reduces the equation for -\( \omega \) to $$ \begin{align} - \boldsymbol{\beta} = \boldsymbol{V}\boldsymbol{\Sigma}^{+} \boldsymbol{U}^T \boldsymbol{y}. -\tag{6} + H = -J \sum_{k}^L s_k s_{k + 1}, +\tag{7} +\end{align} +$$ + +where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins in the system is determined by \( L \). For the one-dimensional system there is no phase transition. + +

    +We will look at a system of \( L = 40 \) spins with a coupling constant of \( J = 1 \). To get enough training data we will generate 10000 states with their respective energies. + +

    + + +

    import numpy as np
    +import matplotlib.pyplot as plt
    +from mpl_toolkits.axes_grid1 import make_axes_locatable
    +import seaborn as sns
    +import scipy.linalg as scl
    +from sklearn.model_selection import train_test_split
    +import sklearn.linear_model as skl
    +import tqdm
    +sns.set(color_codes=True)
    +cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
    +
    +L = 40
    +n = int(1e4)
    +
    +spins = np.random.choice([-1, 1], size=(n, L))
    +J = 1.0
    +
    +energies = np.zeros(n)
    +
    +for i in range(n):
    +    energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
    +
    +

    +A more general form for the one-dimensional Ising model is + +$$ +\begin{align} + H = - \sum_j^L \sum_k^L s_j s_k J_{jk}. +\tag{8} \end{align} $$

    -Note that solving this equation by actually doing the pseudoinverse -(which is what we will do) is not a good idea as this operation scales -as \( \mathcal{O}(n^3) \), where \( n \) is the number of elements in a -general matrix. Instead, doing \( QR \)-factorization and solving the -linear system as an equation would reduce this down to -\( \mathcal{O}(n^2) \) operations. +Here we allow for interactions beyond the nearest neighbors and a more +adaptive coupling matrix. This latter expression can be formulated as +a matrix-product on the form +$$ +\begin{align} + H = X J, +\tag{9} +\end{align} +$$ + +

    +where \( X_{jk} = s_j s_k \) and \( J \) is the matrix consisting of the +elements \( -J_{jk} \). This form of writing the energy fits perfectly +with the form utilized in linear regression, viz. +$$ +\begin{align} + \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}. +\tag{10} +\end{align} +$$ + +We organize the data as we did above +

    + + +

    X = np.zeros((n, L ** 2))
    +for i in range(n):
    +    X[i] = np.outer(spins[i], spins[i]).ravel()
    +y = energies
    +X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.96)
    +
    +X_train_own = np.concatenate(
    +    (np.ones(len(X_train))[:, np.newaxis], X_train),
    +    axis=1
    +)
    +
    +X_test_own = np.concatenate(
    +    (np.ones(len(X_test))[:, np.newaxis], X_test),
    +    axis=1
    +)
    +
    +

    +We will do all fitting with Scikit-Learn,

    -

    def ols_svd(x: np.ndarray, y: np.ndarray) -> np.ndarray:
    -    u, s, v = scl.svd(x)
    -    return v.T @ scl.pinv(scl.diagsvd(s, u.shape[0], v.shape[0])) @ u.T @ y
    +
    clf = skl.LinearRegression().fit(X_train, y_train)
     

    +When extracting the \( J \)-matrix we make sure to remove the intercept +

    -

    beta = ols_svd(X_train_own,y_train)
    +
    J_sk = clf.coef_.reshape(L, L)
     

    -When extracting the \( J \)-matrix we need to make sure that we remove the intercept, as is done here - -

    - - -

    J = beta[1:].reshape(L, L)
    -
    -

    -A way of looking at the coefficients in \( J \) is to plot the matrices as images. - +And then we plot the results

    fig = plt.figure(figsize=(20, 14))
    -im = plt.imshow(J, **cmap_args)
    -plt.title("OLS", fontsize=18)
    +im = plt.imshow(J_sk, **cmap_args)
    +plt.title("LinearRegression from Scikit-learn", fontsize=18)
     plt.xticks(fontsize=18)
     plt.yticks(fontsize=18)
     cb = fig.colorbar(im)
    @@ -498,14 +547,7 @@ cb.ax.se
     plt.show()
     

    -It is interesting to note that OLS -considers both \( J_{j, j + 1} = -0.5 \) and \( J_{j, j - 1} = -0.5 \) as -valid matrix elements for \( J \). -In our discussion below on hyperparameters and Ridge and Lasso regression we will see that -this problem can be removed, partly and only with Lasso regression. - -

    -In this case our matrix inversion was actually possible. The obvious question now is what is the mathematics behind the SVD? +The results perfectly with our previous discussion where we used our own code.

    @@ -533,7 +575,7 @@ In this case our matrix inversion was actually possible. The obvious question no

  • 30
  • 31
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs022.html b/doc/pub/week38/html/._week38-bs022.html index f69710e86..d1b4e1c9a 100644 --- a/doc/pub/week38/html/._week38-bs022.html +++ b/doc/pub/week38/html/._week38-bs022.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,137 +418,38 @@ MathJax.Hub.Config({ -

    The one-dimensional Ising model

    +

    Ridge regression

    -Let us bring back the Ising model again, but now with an additional -focus on Ridge and Lasso regression as well. We repeat some of the -basic parts of the Ising model and the setup of the training and test -data. The one-dimensional Ising model with nearest neighbor -interaction, no external field and a constant coupling constant \( J \) is -given by +Having explored the ordinary least squares we move on to ridge +regression. In ridge regression we include a regularizer. This +involves a new cost function which leads to a new estimate for the +weights \( \boldsymbol{\beta} \). This results in a penalized regression problem. The +cost function is given by $$ \begin{align} - H = -J \sum_{k}^L s_k s_{k + 1}, -\tag{7} -\end{align} -$$ - -where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins in the system is determined by \( L \). For the one-dimensional system there is no phase transition. - -

    -We will look at a system of \( L = 40 \) spins with a coupling constant of \( J = 1 \). To get enough training data we will generate 10000 states with their respective energies. - -

    - - -

    import numpy as np
    -import matplotlib.pyplot as plt
    -from mpl_toolkits.axes_grid1 import make_axes_locatable
    -import seaborn as sns
    -import scipy.linalg as scl
    -from sklearn.model_selection import train_test_split
    -import sklearn.linear_model as skl
    -import tqdm
    -sns.set(color_codes=True)
    -cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
    -
    -L = 40
    -n = int(1e4)
    -
    -spins = np.random.choice([-1, 1], size=(n, L))
    -J = 1.0
    -
    -energies = np.zeros(n)
    -
    -for i in range(n):
    -    energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
    -
    -

    -A more general form for the one-dimensional Ising model is - -$$ -\begin{align} - H = - \sum_j^L \sum_k^L s_j s_k J_{jk}. -\tag{8} + C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta}. +\tag{11} \end{align} $$ -

    -Here we allow for interactions beyond the nearest neighbors and a more -adaptive coupling matrix. This latter expression can be formulated as -a matrix-product on the form -$$ -\begin{align} - H = X J, -\tag{9} -\end{align} -$$ - -

    -where \( X_{jk} = s_j s_k \) and \( J \) is the matrix consisting of the -elements \( -J_{jk} \). This form of writing the energy fits perfectly -with the form utilized in linear regression, viz. -$$ -\begin{align} - \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}. -\tag{10} -\end{align} -$$ - -We organize the data as we did above

    -

    X = np.zeros((n, L ** 2))
    -for i in range(n):
    -    X[i] = np.outer(spins[i], spins[i]).ravel()
    -y = energies
    -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.96)
    -
    -X_train_own = np.concatenate(
    -    (np.ones(len(X_train))[:, np.newaxis], X_train),
    -    axis=1
    -)
    -
    -X_test_own = np.concatenate(
    -    (np.ones(len(X_test))[:, np.newaxis], X_test),
    -    axis=1
    -)
    -
    -

    -We will do all fitting with Scikit-Learn, - -

    - - -

    clf = skl.LinearRegression().fit(X_train, y_train)
    -
    -

    -When extracting the \( J \)-matrix we make sure to remove the intercept -

    - - -

    J_sk = clf.coef_.reshape(L, L)
    -
    -

    -And then we plot the results -

    - - -

    fig = plt.figure(figsize=(20, 14))
    -im = plt.imshow(J_sk, **cmap_args)
    -plt.title("LinearRegression from Scikit-learn", fontsize=18)
    +
    _lambda = 0.1
    +clf_ridge = skl.Ridge(alpha=_lambda).fit(X_train, y_train)
    +J_ridge_sk = clf_ridge.coef_.reshape(L, L)
    +fig = plt.figure(figsize=(20, 14))
    +im = plt.imshow(J_ridge_sk, **cmap_args)
    +plt.title("Ridge from Scikit-learn", fontsize=18)
     plt.xticks(fontsize=18)
     plt.yticks(fontsize=18)
     cb = fig.colorbar(im)
     cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
    +
     plt.show()
     
    -

    -The results perfectly with our previous discussion where we used our own code. -

    @@ -580,7 +476,7 @@ The results perfectly with our previous discussion where we used our own code.

  • 31
  • 32
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs023.html b/doc/pub/week38/html/._week38-bs023.html index 7cf5de9eb..c7b61759f 100644 --- a/doc/pub/week38/html/._week38-bs023.html +++ b/doc/pub/week38/html/._week38-bs023.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,31 +418,29 @@ MathJax.Hub.Config({ -

    Ridge regression

    +

    LASSO regression

    -Having explored the ordinary least squares we move on to ridge -regression. In ridge regression we include a regularizer. This -involves a new cost function which leads to a new estimate for the -weights \( \boldsymbol{\beta} \). This results in a penalized regression problem. The -cost function is given by +In the Least Absolute Shrinkage and Selection Operator (LASSO)-method we get a third cost function. $$ \begin{align} - C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta}. -\tag{11} + C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \sqrt{\boldsymbol{\beta}^T\boldsymbol{\beta}}. +\tag{12} \end{align} $$ +

    +Finding the extremal point of this cost function is not so straight-forward as in least squares and ridge. We will therefore rely solely on the function ``Lasso`` from Scikit-Learn. +

    -

    _lambda = 0.1
    -clf_ridge = skl.Ridge(alpha=_lambda).fit(X_train, y_train)
    -J_ridge_sk = clf_ridge.coef_.reshape(L, L)
    +
    clf_lasso = skl.Lasso(alpha=_lambda).fit(X_train, y_train)
    +J_lasso_sk = clf_lasso.coef_.reshape(L, L)
     fig = plt.figure(figsize=(20, 14))
    -im = plt.imshow(J_ridge_sk, **cmap_args)
    -plt.title("Ridge from Scikit-learn", fontsize=18)
    +im = plt.imshow(J_lasso_sk, **cmap_args)
    +plt.title("Lasso from Scikit-learn", fontsize=18)
     plt.xticks(fontsize=18)
     plt.yticks(fontsize=18)
     cb = fig.colorbar(im)
    @@ -455,6 +448,11 @@ cb.ax.se
     
     plt.show()
     
    +

    +It is quite striking how LASSO breaks the symmetry of the coupling +constant as opposed to ridge and OLS. We get a sparse solution with +\( J_{j, j + 1} = -1 \). +

    @@ -481,7 +479,7 @@ plt.show()

  • 32
  • 33
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs024.html b/doc/pub/week38/html/._week38-bs024.html index df865fe1c..7ac38c92b 100644 --- a/doc/pub/week38/html/._week38-bs024.html +++ b/doc/pub/week38/html/._week38-bs024.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,40 +418,56 @@ MathJax.Hub.Config({ -

    LASSO regression

    +

    Performance as function of the regularization parameter

    -In the Least Absolute Shrinkage and Selection Operator (LASSO)-method we get a third cost function. - -$$ -\begin{align} - C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \sqrt{\boldsymbol{\beta}^T\boldsymbol{\beta}}. -\tag{12} -\end{align} -$$ - -

    -Finding the extremal point of this cost function is not so straight-forward as in least squares and ridge. We will therefore rely solely on the function ``Lasso`` from Scikit-Learn. +We see how the different models perform for a different set of values for \( \lambda \).

    -

    clf_lasso = skl.Lasso(alpha=_lambda).fit(X_train, y_train)
    -J_lasso_sk = clf_lasso.coef_.reshape(L, L)
    -fig = plt.figure(figsize=(20, 14))
    -im = plt.imshow(J_lasso_sk, **cmap_args)
    -plt.title("Lasso from Scikit-learn", fontsize=18)
    -plt.xticks(fontsize=18)
    -plt.yticks(fontsize=18)
    -cb = fig.colorbar(im)
    -cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
    +
    lambdas = np.logspace(-4, 5, 10)
    +
    +train_errors = {
    +    "ols_sk": np.zeros(lambdas.size),
    +    "ridge_sk": np.zeros(lambdas.size),
    +    "lasso_sk": np.zeros(lambdas.size)
    +}
    +
    +test_errors = {
    +    "ols_sk": np.zeros(lambdas.size),
    +    "ridge_sk": np.zeros(lambdas.size),
    +    "lasso_sk": np.zeros(lambdas.size)
    +}
    +
    +plot_counter = 1
    +
    +fig = plt.figure(figsize=(32, 54))
    +
    +for i, _lambda in enumerate(tqdm.tqdm(lambdas)):
    +    for key, method in zip(
    +        ["ols_sk", "ridge_sk", "lasso_sk"],
    +        [skl.LinearRegression(), skl.Ridge(alpha=_lambda), skl.Lasso(alpha=_lambda)]
    +    ):
    +        method = method.fit(X_train, y_train)
    +
    +        train_errors[key][i] = method.score(X_train, y_train)
    +        test_errors[key][i] = method.score(X_test, y_test)
    +
    +        omega = method.coef_.reshape(L, L)
    +
    +        plt.subplot(10, 5, plot_counter)
    +        plt.imshow(omega, **cmap_args)
    +        plt.title(r"%s, $\lambda = %.4f$" % (key, _lambda))
    +        plot_counter += 1
     
     plt.show()
     

    -It is quite striking how LASSO breaks the symmetry of the coupling -constant as opposed to ridge and OLS. We get a sparse solution with -\( J_{j, j + 1} = -1 \). +We see that LASSO reaches a good solution for low +values of \( \lambda \), but will "wither" when we increase \( \lambda \) too +much. Ridge is more stable over a larger range of values for +\( \lambda \), but eventually also fades away.

    @@ -484,7 +495,7 @@ constant as opposed to ridge and OLS. We get a sparse solution with

  • 33
  • 34
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs025.html b/doc/pub/week38/html/._week38-bs025.html index 02f3d0474..3b5c8f105 100644 --- a/doc/pub/week38/html/._week38-bs025.html +++ b/doc/pub/week38/html/._week38-bs025.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,56 +418,54 @@ MathJax.Hub.Config({ -

    Performance as function of the regularization parameter

    +

    Finding the optimal value of \( \lambda \)

    -We see how the different models perform for a different set of values for \( \lambda \). +To determine which value of \( \lambda \) is best we plot the accuracy of +the models when predicting the training and the testing set. We expect +the accuracy of the training set to be quite good, but if the accuracy +of the testing set is much lower this tells us that we might be +subject to an overfit model. The ideal scenario is an accuracy on the +testing set that is close to the accuracy of the training set.

    -

    lambdas = np.logspace(-4, 5, 10)
    +
    fig = plt.figure(figsize=(20, 14))
     
    -train_errors = {
    -    "ols_sk": np.zeros(lambdas.size),
    -    "ridge_sk": np.zeros(lambdas.size),
    -    "lasso_sk": np.zeros(lambdas.size)
    +colors = {
    +    "ols_sk": "r",
    +    "ridge_sk": "y",
    +    "lasso_sk": "c"
     }
     
    -test_errors = {
    -    "ols_sk": np.zeros(lambdas.size),
    -    "ridge_sk": np.zeros(lambdas.size),
    -    "lasso_sk": np.zeros(lambdas.size)
    -}
    -
    -plot_counter = 1
    -
    -fig = plt.figure(figsize=(32, 54))
    -
    -for i, _lambda in enumerate(tqdm.tqdm(lambdas)):
    -    for key, method in zip(
    -        ["ols_sk", "ridge_sk", "lasso_sk"],
    -        [skl.LinearRegression(), skl.Ridge(alpha=_lambda), skl.Lasso(alpha=_lambda)]
    -    ):
    -        method = method.fit(X_train, y_train)
    -
    -        train_errors[key][i] = method.score(X_train, y_train)
    -        test_errors[key][i] = method.score(X_test, y_test)
    -
    -        omega = method.coef_.reshape(L, L)
    -
    -        plt.subplot(10, 5, plot_counter)
    -        plt.imshow(omega, **cmap_args)
    -        plt.title(r"%s, $\lambda = %.4f$" % (key, _lambda))
    -        plot_counter += 1
    +for key in train_errors:
    +    plt.semilogx(
    +        lambdas,
    +        train_errors[key],
    +        colors[key],
    +        label="Train {0}".format(key),
    +        linewidth=4.0
    +    )
     
    +for key in test_errors:
    +    plt.semilogx(
    +        lambdas,
    +        test_errors[key],
    +        colors[key] + "--",
    +        label="Test {0}".format(key),
    +        linewidth=4.0
    +    )
    +plt.legend(loc="best", fontsize=18)
    +plt.xlabel(r"$\lambda$", fontsize=18)
    +plt.ylabel(r"$R^2$", fontsize=18)
    +plt.tick_params(labelsize=18)
     plt.show()
     

    -We see that LASSO reaches a good solution for low -values of \( \lambda \), but will "wither" when we increase \( \lambda \) too -much. Ridge is more stable over a larger range of values for -\( \lambda \), but eventually also fades away. +From the above figure we can see that LASSO with \( \lambda = 10^{-2} \) +achieves a very good accuracy on the test set. This by far surpasses the +other models for all values of \( \lambda \).

    @@ -500,7 +493,7 @@ much. Ridge is more stable over a larger range of values for

  • 34
  • 35
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs026.html b/doc/pub/week38/html/._week38-bs026.html index 79e3730df..8019c1d72 100644 --- a/doc/pub/week38/html/._week38-bs026.html +++ b/doc/pub/week38/html/._week38-bs026.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -421,56 +416,22 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Finding the optimal value of \( \lambda \)

    +

    Logistic Regression

    -To determine which value of \( \lambda \) is best we plot the accuracy of -the models when predicting the training and the testing set. We expect -the accuracy of the training set to be quite good, but if the accuracy -of the testing set is much lower this tells us that we might be -subject to an overfit model. The ideal scenario is an accuracy on the -testing set that is close to the accuracy of the training set. - -

    - - -

    fig = plt.figure(figsize=(20, 14))
    -
    -colors = {
    -    "ols_sk": "r",
    -    "ridge_sk": "y",
    -    "lasso_sk": "c"
    -}
    -
    -for key in train_errors:
    -    plt.semilogx(
    -        lambdas,
    -        train_errors[key],
    -        colors[key],
    -        label="Train {0}".format(key),
    -        linewidth=4.0
    -    )
    -
    -for key in test_errors:
    -    plt.semilogx(
    -        lambdas,
    -        test_errors[key],
    -        colors[key] + "--",
    -        label="Test {0}".format(key),
    -        linewidth=4.0
    -    )
    -plt.legend(loc="best", fontsize=18)
    -plt.xlabel(r"$\lambda$", fontsize=18)
    -plt.ylabel(r"$R^2$", fontsize=18)
    -plt.tick_params(labelsize=18)
    -plt.show()
    -
    -

    -From the above figure we can see that LASSO with \( \lambda = 10^{-2} \) -achieves a very good accuracy on the test set. This by far surpasses the -other models for all values of \( \lambda \). +In linear regression our main interest was centered on learning the +coefficients of a functional fit (say a polynomial) in order to be +able to predict the response of a continuous variable on some unseen +data. The fit to the continuous variable \( y_i \) is based on some +independent variables \( \boldsymbol{x}_i \). Linear regression resulted in +analytical expressions for standard ordinary Least Squares or Ridge +regression (in terms of matrices to invert) for several quantities, +ranging from the variance and thereby the confidence intervals of the +parameters \( \boldsymbol{\beta} \) to the mean squared error. If we can invert +the product of the design matrices, linear regression gives then a +simple recipe for fitting our data.

    @@ -498,7 +459,7 @@ other models for all values of \( \lambda \).

  • 35
  • 36
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs027.html b/doc/pub/week38/html/._week38-bs027.html index f18d73644..ec191791b 100644 --- a/doc/pub/week38/html/._week38-bs027.html +++ b/doc/pub/week38/html/._week38-bs027.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,20 +418,25 @@ MathJax.Hub.Config({ -

    Logistic Regression

    +

    Classification problems

    -In linear regression our main interest was centered on learning the -coefficients of a functional fit (say a polynomial) in order to be -able to predict the response of a continuous variable on some unseen -data. The fit to the continuous variable \( y_i \) is based on some -independent variables \( \boldsymbol{x}_i \). Linear regression resulted in -analytical expressions for standard ordinary Least Squares or Ridge -regression (in terms of matrices to invert) for several quantities, -ranging from the variance and thereby the confidence intervals of the -parameters \( \boldsymbol{\beta} \) to the mean squared error. If we can invert -the product of the design matrices, linear regression gives then a -simple recipe for fitting our data. +Classification problems, however, are concerned with outcomes taking +the form of discrete variables (i.e. categories). We may for example, +on the basis of DNA sequencing for a number of patients, like to find +out which mutations are important for a certain disease; or based on +scans of various patients' brains, figure out if there is a tumor or +not; or given a specific physical system, we'd like to identify its +state, say whether it is an ordered or disordered system (typical +situation in solid state physics); or classify the status of a +patient, whether she/he has a stroke or not and many other similar +situations. + +

    +The most common situation we encounter when we apply logistic +regression is that of two possible outcomes, normally denoted as a +binary outcome, true or false, positive or negative, success or +failure etc.

    @@ -464,7 +464,7 @@ simple recipe for fitting our data.

  • 36
  • 37
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs028.html b/doc/pub/week38/html/._week38-bs028.html index 6d5f0796b..c17ab46b3 100644 --- a/doc/pub/week38/html/._week38-bs028.html +++ b/doc/pub/week38/html/._week38-bs028.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -421,27 +416,25 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Classification problems

    +

    Optimization and Deep learning

    -Classification problems, however, are concerned with outcomes taking -the form of discrete variables (i.e. categories). We may for example, -on the basis of DNA sequencing for a number of patients, like to find -out which mutations are important for a certain disease; or based on -scans of various patients' brains, figure out if there is a tumor or -not; or given a specific physical system, we'd like to identify its -state, say whether it is an ordered or disordered system (typical -situation in solid state physics); or classify the status of a -patient, whether she/he has a stroke or not and many other similar -situations. +Logistic regression will also serve as our stepping stone towards +neural network algorithms and supervised deep learning. For logistic +learning, the minimization of the cost function leads to a non-linear +equation in the parameters \( \boldsymbol{\beta} \). The optimization of the +problem calls therefore for minimization algorithms. This forms the +bottle neck of all machine learning algorithms, namely how to find +reliable minima of a multi-variable function. This leads us to the +family of gradient descent methods. The latter are the working horses +of basically all modern machine learning algorithms.

    -The most common situation we encounter when we apply logistic -regression is that of two possible outcomes, normally denoted as a -binary outcome, true or false, positive or negative, success or -failure etc. +We note also that many of the topics discussed here on logistic +regression are also commonly used in modern supervised Deep Learning +models, as we will see later.

    @@ -469,7 +462,7 @@ failure etc.

  • 37
  • 38
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs029.html b/doc/pub/week38/html/._week38-bs029.html index d2e947ca1..1633e4aa3 100644 --- a/doc/pub/week38/html/._week38-bs029.html +++ b/doc/pub/week38/html/._week38-bs029.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -421,25 +416,31 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Optimization and Deep learning

    +

    Basics

    -Logistic regression will also serve as our stepping stone towards -neural network algorithms and supervised deep learning. For logistic -learning, the minimization of the cost function leads to a non-linear -equation in the parameters \( \boldsymbol{\beta} \). The optimization of the -problem calls therefore for minimization algorithms. This forms the -bottle neck of all machine learning algorithms, namely how to find -reliable minima of a multi-variable function. This leads us to the -family of gradient descent methods. The latter are the working horses -of basically all modern machine learning algorithms. +We consider the case where the dependent variables, also called the +responses or the outcomes, \( y_i \) are discrete and only take values +from \( k=0,\dots,K-1 \) (i.e. \( K \) classes).

    -We note also that many of the topics discussed here on logistic -regression are also commonly used in modern supervised Deep Learning -models, as we will see later. +The goal is to predict the +output classes from the design matrix \( \boldsymbol{X}\in\mathbb{R}^{n\times p} \) +made of \( n \) samples, each of which carries \( p \) features or predictors. The +primary goal is to identify the classes to which new unseen samples +belong. + +

    +Let us specialize to the case of two classes only, with outputs +\( y_i=0 \) and \( y_i=1 \). Our outcomes could represent the status of a +credit card user that could default or not on her/his credit card +debt. That is + +$$ +y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}. +$$

    @@ -467,7 +468,7 @@ models, as we will see later.

  • 38
  • 39
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs030.html b/doc/pub/week38/html/._week38-bs030.html index 25f02bff3..ecae8af76 100644 --- a/doc/pub/week38/html/._week38-bs030.html +++ b/doc/pub/week38/html/._week38-bs030.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -421,32 +416,29 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Basics

    +

    Linear classifier

    -We consider the case where the dependent variables, also called the -responses or the outcomes, \( y_i \) are discrete and only take values -from \( k=0,\dots,K-1 \) (i.e. \( K \) classes). +Before moving to the logistic model, let us try to use our linear +regression model to classify these two outcomes. We could for example +fit a linear model to the default case if \( y_i > 0.5 \) and the no +default case \( y_i \leq 0.5 \).

    -The goal is to predict the -output classes from the design matrix \( \boldsymbol{X}\in\mathbb{R}^{n\times p} \) -made of \( n \) samples, each of which carries \( p \) features or predictors. The -primary goal is to identify the classes to which new unseen samples -belong. - -

    -Let us specialize to the case of two classes only, with outputs -\( y_i=0 \) and \( y_i=1 \). Our outcomes could represent the status of a -credit card user that could default or not on her/his credit card -debt. That is - +We would then have our +weighted linear combination, namely $$ -y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}. +\begin{equation} +\boldsymbol{y} = \boldsymbol{X}^T\boldsymbol{\beta} + \boldsymbol{\epsilon}, +\tag{13} +\end{equation} $$ +where \( \boldsymbol{y} \) is a vector representing the possible outcomes, \( \boldsymbol{X} \) is our +\( n\times p \) design matrix and \( \boldsymbol{\beta} \) represents our estimators/predictors. +

    @@ -473,7 +465,7 @@ $$

  • 39
  • 40
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs031.html b/doc/pub/week38/html/._week38-bs031.html index 5e191feaf..806addd4a 100644 --- a/doc/pub/week38/html/._week38-bs031.html +++ b/doc/pub/week38/html/._week38-bs031.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,26 +418,24 @@ MathJax.Hub.Config({ -

    Linear classifier

    +

    Some selected properties

    -Before moving to the logistic model, let us try to use our linear -regression model to classify these two outcomes. We could for example -fit a linear model to the default case if \( y_i > 0.5 \) and the no -default case \( y_i \leq 0.5 \). +The main problem with our function is that it takes values on the +entire real axis. In the case of logistic regression, however, the +labels \( y_i \) are discrete variables. A typical example is the credit +card data discussed below here, where we can set the state of +defaulting the debt to \( y_i=1 \) and not to \( y_i=0 \) for one the persons +in the data set (see the full example below).

    -We would then have our -weighted linear combination, namely -$$ -\begin{equation} -\boldsymbol{y} = \boldsymbol{X}^T\boldsymbol{\beta} + \boldsymbol{\epsilon}, -\tag{13} -\end{equation} -$$ - -where \( \boldsymbol{y} \) is a vector representing the possible outcomes, \( \boldsymbol{X} \) is our -\( n\times p \) design matrix and \( \boldsymbol{\beta} \) represents our estimators/predictors. +One simple way to get a discrete output is to have sign +functions that map the output of a linear regressor to values \( \{0,1\} \), +\( f(s_i)=sign(s_i)=1 \) if \( s_i\ge 0 \) and 0 if otherwise. +We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron" model in the machine learning +literature. This model is extremely simple. However, in many cases it is more +favorable to use a ``soft" classifier that outputs +the probability of a given category. This leads us to the logistic function.

    @@ -470,7 +463,7 @@ where \( \boldsymbol{y} \) is a vector representing the possible outcomes, \( \b

  • 40
  • 41
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs032.html b/doc/pub/week38/html/._week38-bs032.html index 784425453..d4bbd8380 100644 --- a/doc/pub/week38/html/._week38-bs032.html +++ b/doc/pub/week38/html/._week38-bs032.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,25 +418,69 @@ MathJax.Hub.Config({ -

    Some selected properties

    +

    Simple example

    -The main problem with our function is that it takes values on the -entire real axis. In the case of logistic regression, however, the -labels \( y_i \) are discrete variables. A typical example is the credit -card data discussed below here, where we can set the state of -defaulting the debt to \( y_i=1 \) and not to \( y_i=0 \) for one the persons -in the data set (see the full example below). +The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful.

    -One simple way to get a discrete output is to have sign -functions that map the output of a linear regressor to values \( \{0,1\} \), -\( f(s_i)=sign(s_i)=1 \) if \( s_i\ge 0 \) and 0 if otherwise. -We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron" model in the machine learning -literature. This model is extremely simple. However, in many cases it is more -favorable to use a ``soft" classifier that outputs -the probability of a given category. This leads us to the logistic function. + +

    # Common imports
    +import os
    +import numpy as np
    +import pandas as pd
    +import matplotlib.pyplot as plt
    +from sklearn.linear_model import LinearRegression, Ridge, Lasso
    +from sklearn.model_selection import train_test_split
    +from sklearn.utils import resample
    +from sklearn.metrics import mean_squared_error
    +from IPython.display import display
    +from pylab import plt, mpl
    +plt.style.use('seaborn')
    +mpl.rcParams['font.family'] = 'serif'
    +
    +# Where to save the figures and data files
    +PROJECT_ROOT_DIR = "Results"
    +FIGURE_ID = "Results/FigureFiles"
    +DATA_ID = "DataFiles/"
    +
    +if not os.path.exists(PROJECT_ROOT_DIR):
    +    os.mkdir(PROJECT_ROOT_DIR)
    +
    +if not os.path.exists(FIGURE_ID):
    +    os.makedirs(FIGURE_ID)
    +
    +if not os.path.exists(DATA_ID):
    +    os.makedirs(DATA_ID)
    +
    +def image_path(fig_id):
    +    return os.path.join(FIGURE_ID, fig_id)
    +
    +def data_path(dat_id):
    +    return os.path.join(DATA_ID, dat_id)
    +
    +def save_fig(fig_id):
    +    plt.savefig(image_path(fig_id) + ".png", format='png')
    +
    +infile = open(data_path("chddata.csv"),'r')
    +
    +# Read the chd data as  csv file and organize the data into arrays with age group, age, and chd
    +chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))
    +chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']
    +output = chd['CHD']
    +age = chd['Age']
    +agegroup = chd['Agegroup']
    +numberID  = chd['ID'] 
    +display(chd)
    +
    +plt.scatter(age, output, marker='o')
    +plt.axis([18,70.0,-0.1, 1.2])
    +plt.xlabel(r'Age')
    +plt.ylabel(r'CHD')
    +plt.title(r'Age distribution and Coronary heart disease')
    +plt.show()
    +

    @@ -468,7 +507,7 @@ the probability of a given category. This leads us to the logistic function.

  • 41
  • 42
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs033.html b/doc/pub/week38/html/._week38-bs033.html index a22a99d32..aae61091f 100644 --- a/doc/pub/week38/html/._week38-bs033.html +++ b/doc/pub/week38/html/._week38-bs033.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,69 +418,42 @@ MathJax.Hub.Config({ -

    Simple example

    +

    Plotting the mean value for each group

    -The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful. +What we could attempt however is to plot the mean value for each group.

    -

    # Common imports
    -import os
    -import numpy as np
    -import pandas as pd
    -import matplotlib.pyplot as plt
    -from sklearn.linear_model import LinearRegression, Ridge, Lasso
    -from sklearn.model_selection import train_test_split
    -from sklearn.utils import resample
    -from sklearn.metrics import mean_squared_error
    -from IPython.display import display
    -from pylab import plt, mpl
    -plt.style.use('seaborn')
    -mpl.rcParams['font.family'] = 'serif'
    -
    -# Where to save the figures and data files
    -PROJECT_ROOT_DIR = "Results"
    -FIGURE_ID = "Results/FigureFiles"
    -DATA_ID = "DataFiles/"
    -
    -if not os.path.exists(PROJECT_ROOT_DIR):
    -    os.mkdir(PROJECT_ROOT_DIR)
    -
    -if not os.path.exists(FIGURE_ID):
    -    os.makedirs(FIGURE_ID)
    -
    -if not os.path.exists(DATA_ID):
    -    os.makedirs(DATA_ID)
    -
    -def image_path(fig_id):
    -    return os.path.join(FIGURE_ID, fig_id)
    -
    -def data_path(dat_id):
    -    return os.path.join(DATA_ID, dat_id)
    -
    -def save_fig(fig_id):
    -    plt.savefig(image_path(fig_id) + ".png", format='png')
    -
    -infile = open(data_path("chddata.csv"),'r')
    -
    -# Read the chd data as  csv file and organize the data into arrays with age group, age, and chd
    -chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))
    -chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']
    -output = chd['CHD']
    -age = chd['Age']
    -agegroup = chd['Agegroup']
    -numberID  = chd['ID'] 
    -display(chd)
    -
    -plt.scatter(age, output, marker='o')
    -plt.axis([18,70.0,-0.1, 1.2])
    -plt.xlabel(r'Age')
    -plt.ylabel(r'CHD')
    -plt.title(r'Age distribution and Coronary heart disease')
    +
    agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])
    +group = np.array([1, 2, 3, 4, 5, 6, 7, 8])
    +plt.plot(group, agegroupmean, "r-")
    +plt.axis([0,9,0, 1.0])
    +plt.xlabel(r'Age group')
    +plt.ylabel(r'CHD mean values')
    +plt.title(r'Mean values for each age group')
     plt.show()
     
    +

    +We are now trying to find a function \( f(y\vert x) \), that is a function which gives us an expected value for the output \( y \) with a given input \( x \). +In standard linear regression with a linear dependence on \( x \), we would write this in terms of our model +$$ +f(y_i\vert x_i)=\beta_0+\beta_1 x_i. +$$ + +

    +This expression implies however that \( f(y_i\vert x_i) \) could take any +value from minus infinity to plus infinity. If we however let +\( f(y\vert y) \) be represented by the mean value, the above example +shows us that we can constrain the function to take values between +zero and one, that is we have \( 0 \le f(y_i\vert x_i) \le 1 \). Looking +at our last curve we see also that it has an S-shaped form. This leads +us to a very popular model for the function \( f \), namely the so-called +Sigmoid function or logistic model. We will consider this function as +representing the probability for finding a value of \( y_i \) with a given +\( x_i \). +

    @@ -512,7 +480,7 @@ plt.show()

  • 42
  • 43
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs034.html b/doc/pub/week38/html/._week38-bs034.html index b4df67c5e..850014b15 100644 --- a/doc/pub/week38/html/._week38-bs034.html +++ b/doc/pub/week38/html/._week38-bs034.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,41 +418,25 @@ MathJax.Hub.Config({ -

    Plotting the mean value for each group

    +

    The logistic function

    -What we could attempt however is to plot the mean value for each group. - -

    - - -

    agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])
    -group = np.array([1, 2, 3, 4, 5, 6, 7, 8])
    -plt.plot(group, agegroupmean, "r-")
    -plt.axis([0,9,0, 1.0])
    -plt.xlabel(r'Age group')
    -plt.ylabel(r'CHD mean values')
    -plt.title(r'Mean values for each age group')
    -plt.show()
    -
    -

    -We are now trying to find a function \( f(y\vert x) \), that is a function which gives us an expected value for the output \( y \) with a given input \( x \). -In standard linear regression with a linear dependence on \( x \), we would write this in terms of our model +Another widely studied model, is the so-called +perceptron model, which is an example of a "hard classification" model. We +will encounter this model when we discuss neural networks as +well. Each datapoint is deterministically assigned to a category (i.e +\( y_i=0 \) or \( y_i=1 \)). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a "soft" +classifier that outputs the probability of a given category rather +than a single value. For example, given \( x_i \), the classifier +outputs the probability of being in a category \( k \). Logistic regression +is the most common example of a so-called soft classifier. In logistic +regression, the probability that a data point \( x_i \) +belongs to a category \( y_i=\{0,1\} \) is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event, $$ -f(y_i\vert x_i)=\beta_0+\beta_1 x_i. +p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}. $$ -

    -This expression implies however that \( f(y_i\vert x_i) \) could take any -value from minus infinity to plus infinity. If we however let -\( f(y\vert y) \) be represented by the mean value, the above example -shows us that we can constrain the function to take values between -zero and one, that is we have \( 0 \le f(y_i\vert x_i) \le 1 \). Looking -at our last curve we see also that it has an S-shaped form. This leads -us to a very popular model for the function \( f \), namely the so-called -Sigmoid function or logistic model. We will consider this function as -representing the probability for finding a value of \( y_i \) with a given -\( x_i \). +Note that \( 1-p(t)= p(-t) \).

    @@ -485,7 +464,7 @@ representing the probability for finding a value of \( y_i \) with a given

  • 43
  • 44
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs035.html b/doc/pub/week38/html/._week38-bs035.html index e39c002e8..4acbc04d7 100644 --- a/doc/pub/week38/html/._week38-bs035.html +++ b/doc/pub/week38/html/._week38-bs035.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,26 +418,69 @@ MathJax.Hub.Config({ -

    The logistic function

    +

    Examples of likelihood functions used in logistic regression and nueral networks

    -Another widely studied model, is the so-called -perceptron model, which is an example of a "hard classification" model. We -will encounter this model when we discuss neural networks as -well. Each datapoint is deterministically assigned to a category (i.e -\( y_i=0 \) or \( y_i=1 \)). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a "soft" -classifier that outputs the probability of a given category rather -than a single value. For example, given \( x_i \), the classifier -outputs the probability of being in a category \( k \). Logistic regression -is the most common example of a so-called soft classifier. In logistic -regression, the probability that a data point \( x_i \) -belongs to a category \( y_i=\{0,1\} \) is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event, -$$ -p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}. -$$ +The following code plots the logistic function, the step function and other functions we will encounter from here and on. -Note that \( 1-p(t)= p(-t) \). +

    + +

    """The sigmoid function (or the logistic curve) is a
    +function that takes any real number, z, and outputs a number (0,1).
    +It is useful in neural networks for assigning weights on a relative scale.
    +The value z is the weighted sum of parameters involved in the learning algorithm."""
    +
    +import numpy
    +import matplotlib.pyplot as plt
    +import math as mt
    +
    +z = numpy.arange(-5, 5, .1)
    +sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))
    +sigma = sigma_fn(z)
    +
    +fig = plt.figure()
    +ax = fig.add_subplot(111)
    +ax.plot(z, sigma)
    +ax.set_ylim([-0.1, 1.1])
    +ax.set_xlim([-5,5])
    +ax.grid(True)
    +ax.set_xlabel('z')
    +ax.set_title('sigmoid function')
    +
    +plt.show()
    +
    +"""Step Function"""
    +z = numpy.arange(-5, 5, .02)
    +step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)
    +step = step_fn(z)
    +
    +fig = plt.figure()
    +ax = fig.add_subplot(111)
    +ax.plot(z, step)
    +ax.set_ylim([-0.5, 1.5])
    +ax.set_xlim([-5,5])
    +ax.grid(True)
    +ax.set_xlabel('z')
    +ax.set_title('step function')
    +
    +plt.show()
    +
    +"""tanh Function"""
    +z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)
    +t = numpy.tanh(z)
    +
    +fig = plt.figure()
    +ax = fig.add_subplot(111)
    +ax.plot(z, t)
    +ax.set_ylim([-1.0, 1.0])
    +ax.set_xlim([-2*mt.pi,2*mt.pi])
    +ax.grid(True)
    +ax.set_xlabel('z')
    +ax.set_title('tanh function')
    +
    +plt.show()
    +

    @@ -469,7 +507,7 @@ Note that \( 1-p(t)= p(-t) \).

  • 44
  • 45
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs036.html b/doc/pub/week38/html/._week38-bs036.html index a53aa23bb..e6c59bd28 100644 --- a/doc/pub/week38/html/._week38-bs036.html +++ b/doc/pub/week38/html/._week38-bs036.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,69 +418,25 @@ MathJax.Hub.Config({ -

    Examples of likelihood functions used in logistic regression and nueral networks

    +

    Two parameters

    -The following code plots the logistic function, the step function and other functions we will encounter from here and on. +We assume now that we have two classes with \( y_i \) either \( 0 \) or \( 1 \). Furthermore we assume also that we have only two parameters \( \beta \) in our fitting of the Sigmoid function, that is we define probabilities +$$ +\begin{align*} +p(y_i=1|x_i,\boldsymbol{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\ +p(y_i=0|x_i,\boldsymbol{\beta}) &= 1 - p(y_i=1|x_i,\boldsymbol{\beta}), +\end{align*} +$$ + +where \( \boldsymbol{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).

    +Note that we used +$$ +p(y_i=0\vert x_i, \boldsymbol{\beta}) = 1-p(y_i=1\vert x_i, \boldsymbol{\beta}). +$$ - -

    """The sigmoid function (or the logistic curve) is a
    -function that takes any real number, z, and outputs a number (0,1).
    -It is useful in neural networks for assigning weights on a relative scale.
    -The value z is the weighted sum of parameters involved in the learning algorithm."""
    -
    -import numpy
    -import matplotlib.pyplot as plt
    -import math as mt
    -
    -z = numpy.arange(-5, 5, .1)
    -sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))
    -sigma = sigma_fn(z)
    -
    -fig = plt.figure()
    -ax = fig.add_subplot(111)
    -ax.plot(z, sigma)
    -ax.set_ylim([-0.1, 1.1])
    -ax.set_xlim([-5,5])
    -ax.grid(True)
    -ax.set_xlabel('z')
    -ax.set_title('sigmoid function')
    -
    -plt.show()
    -
    -"""Step Function"""
    -z = numpy.arange(-5, 5, .02)
    -step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)
    -step = step_fn(z)
    -
    -fig = plt.figure()
    -ax = fig.add_subplot(111)
    -ax.plot(z, step)
    -ax.set_ylim([-0.5, 1.5])
    -ax.set_xlim([-5,5])
    -ax.grid(True)
    -ax.set_xlabel('z')
    -ax.set_title('step function')
    -
    -plt.show()
    -
    -"""tanh Function"""
    -z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)
    -t = numpy.tanh(z)
    -
    -fig = plt.figure()
    -ax = fig.add_subplot(111)
    -ax.plot(z, t)
    -ax.set_ylim([-1.0, 1.0])
    -ax.set_xlim([-2*mt.pi,2*mt.pi])
    -ax.grid(True)
    -ax.set_xlabel('z')
    -ax.set_title('tanh function')
    -
    -plt.show()
    -

    @@ -512,7 +463,7 @@ plt.show()

  • 45
  • 46
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs037.html b/doc/pub/week38/html/._week38-bs037.html index 0a66ee898..63291841f 100644 --- a/doc/pub/week38/html/._week38-bs037.html +++ b/doc/pub/week38/html/._week38-bs037.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -421,25 +416,26 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Two parameters

    +

    Maximum likelihood

    -We assume now that we have two classes with \( y_i \) either \( 0 \) or \( 1 \). Furthermore we assume also that we have only two parameters \( \beta \) in our fitting of the Sigmoid function, that is we define probabilities +In order to define the total likelihood for all possible outcomes from a +dataset \( \mathcal{D}=\{(y_i,x_i)\} \), with the binary labels +\( y_i\in\{0,1\} \) and where the data points are drawn independently, we use the so-called Maximum Likelihood Estimation (MLE) principle. +We aim thus at maximizing +the probability of seeing the observed data. We can then approximate the +likelihood in terms of the product of the individual probabilities of a specific outcome \( y_i \), that is $$ \begin{align*} -p(y_i=1|x_i,\boldsymbol{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\ -p(y_i=0|x_i,\boldsymbol{\beta}) &= 1 - p(y_i=1|x_i,\boldsymbol{\beta}), +P(\mathcal{D}|\boldsymbol{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\boldsymbol{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\boldsymbol{\beta}))\right]^{1-y_i}\nonumber \\ \end{align*} $$ -where \( \boldsymbol{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \). - -

    -Note that we used +from which we obtain the log-likelihood and our cost/loss function $$ -p(y_i=0\vert x_i, \boldsymbol{\beta}) = 1-p(y_i=1\vert x_i, \boldsymbol{\beta}). +\mathcal{C}(\boldsymbol{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\boldsymbol{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\boldsymbol{\beta}))\right]\right). $$

    @@ -468,7 +464,7 @@ $$

  • 46
  • 47
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs038.html b/doc/pub/week38/html/._week38-bs038.html index fedb97b7a..58c4b249a 100644 --- a/doc/pub/week38/html/._week38-bs038.html +++ b/doc/pub/week38/html/._week38-bs038.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -421,28 +416,26 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Maximum likelihood

    +

    The cost function rewritten

    -In order to define the total likelihood for all possible outcomes from a -dataset \( \mathcal{D}=\{(y_i,x_i)\} \), with the binary labels -\( y_i\in\{0,1\} \) and where the data points are drawn independently, we use the so-called Maximum Likelihood Estimation (MLE) principle. -We aim thus at maximizing -the probability of seeing the observed data. We can then approximate the -likelihood in terms of the product of the individual probabilities of a specific outcome \( y_i \), that is +Reordering the logarithms, we can rewrite the cost/loss function as $$ -\begin{align*} -P(\mathcal{D}|\boldsymbol{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\boldsymbol{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\boldsymbol{\beta}))\right]^{1-y_i}\nonumber \\ -\end{align*} +\mathcal{C}(\boldsymbol{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right). $$ -from which we obtain the log-likelihood and our cost/loss function +

    +The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to \( \beta \). +Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that $$ -\mathcal{C}(\boldsymbol{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\boldsymbol{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\boldsymbol{\beta}))\right]\right). +\mathcal{C}(\boldsymbol{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right). $$ +This equation is known in statistics as the cross entropy. Finally, we note that just as in linear regression, +in practice we often supplement the cross-entropy with additional regularization terms, usually \( L_1 \) and \( L_2 \) regularization as we did for Ridge and Lasso regression. +

    @@ -469,7 +462,7 @@ $$

  • 47
  • 48
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs039.html b/doc/pub/week38/html/._week38-bs039.html index 7d67584f0..e177ad1ab 100644 --- a/doc/pub/week38/html/._week38-bs039.html +++ b/doc/pub/week38/html/._week38-bs039.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,23 +418,24 @@ MathJax.Hub.Config({ -

    The cost function rewritten

    +

    Minimizing the cross entropy

    -Reordering the logarithms, we can rewrite the cost/loss function as -$$ -\mathcal{C}(\boldsymbol{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right). -$$ +The cross entropy is a convex function of the weights \( \boldsymbol{\beta} \) and, +therefore, any local minimizer is a global minimizer.

    -The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to \( \beta \). -Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that +Minimizing this +cost function with respect to the two parameters \( \beta_0 \) and \( \beta_1 \) we obtain + $$ -\mathcal{C}(\boldsymbol{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right). +\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right), $$ -This equation is known in statistics as the cross entropy. Finally, we note that just as in linear regression, -in practice we often supplement the cross-entropy with additional regularization terms, usually \( L_1 \) and \( L_2 \) regularization as we did for Ridge and Lasso regression. +and +$$ +\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right). +$$

    @@ -467,7 +463,7 @@ in practice we often supplement the cross-entropy with additional regularization

  • 48
  • 49
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs040.html b/doc/pub/week38/html/._week38-bs040.html index 0f0d15c2d..6010937b7 100644 --- a/doc/pub/week38/html/._week38-bs040.html +++ b/doc/pub/week38/html/._week38-bs040.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,23 +418,24 @@ MathJax.Hub.Config({ -

    Minimizing the cross entropy

    +

    A more compact expression

    -The cross entropy is a convex function of the weights \( \boldsymbol{\beta} \) and, -therefore, any local minimizer is a global minimizer. +Let us now define a vector \( \boldsymbol{y} \) with \( n \) elements \( y_i \), an +\( n\times p \) matrix \( \boldsymbol{X} \) which contains the \( x_i \) values and a +vector \( \boldsymbol{p} \) of fitted probabilities \( p(y_i\vert x_i,\boldsymbol{\beta}) \). We can rewrite in a more compact form the first +derivative of cost function as + +$$ +\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{p}\right). +$$

    -Minimizing this -cost function with respect to the two parameters \( \beta_0 \) and \( \beta_1 \) we obtain +If we in addition define a diagonal matrix \( \boldsymbol{W} \) with elements +\( p(y_i\vert x_i,\boldsymbol{\beta})(1-p(y_i\vert x_i,\boldsymbol{\beta}) \), we can obtain a compact expression of the second derivative as $$ -\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right), -$$ - -and -$$ -\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right). +\frac{\partial^2 \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}\partial \boldsymbol{\beta}^T} = \boldsymbol{X}^T\boldsymbol{W}\boldsymbol{X}. $$

    @@ -468,7 +464,7 @@ $$

  • 49
  • 50
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs041.html b/doc/pub/week38/html/._week38-bs041.html index ba527bc21..1024a0957 100644 --- a/doc/pub/week38/html/._week38-bs041.html +++ b/doc/pub/week38/html/._week38-bs041.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,24 +418,17 @@ MathJax.Hub.Config({ -

    A more compact expression

    +

    Extending to more predictors

    -Let us now define a vector \( \boldsymbol{y} \) with \( n \) elements \( y_i \), an -\( n\times p \) matrix \( \boldsymbol{X} \) which contains the \( x_i \) values and a -vector \( \boldsymbol{p} \) of fitted probabilities \( p(y_i\vert x_i,\boldsymbol{\beta}) \). We can rewrite in a more compact form the first -derivative of cost function as - +Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with \( p \) predictors $$ -\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{p}\right). +\log{ \frac{p(\boldsymbol{\beta}\boldsymbol{x})}{1-p(\boldsymbol{\beta}\boldsymbol{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p. $$ -

    -If we in addition define a diagonal matrix \( \boldsymbol{W} \) with elements -\( p(y_i\vert x_i,\boldsymbol{\beta})(1-p(y_i\vert x_i,\boldsymbol{\beta}) \), we can obtain a compact expression of the second derivative as - +Here we defined \( \boldsymbol{x}=[1,x_1,x_2,\dots,x_p] \) and \( \boldsymbol{\beta}=[\beta_0, \beta_1, \dots, \beta_p] \) leading to $$ -\frac{\partial^2 \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}\partial \boldsymbol{\beta}^T} = \boldsymbol{X}^T\boldsymbol{W}\boldsymbol{X}. +p(\boldsymbol{\beta}\boldsymbol{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}. $$

    @@ -469,7 +457,7 @@ $$

  • 50
  • 51
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/._week38-bs042.html b/doc/pub/week38/html/._week38-bs042.html index df0ea302a..6e8f03803 100644 --- a/doc/pub/week38/html/._week38-bs042.html +++ b/doc/pub/week38/html/._week38-bs042.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -423,19 +418,31 @@ MathJax.Hub.Config({ -

    Extending to more predictors

    +

    Including more classes

    -Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with \( p \) predictors +Till now we have mainly focused on two classes, the so-called binary +system. Suppose we wish to extend to \( K \) classes. Let us for the sake +of simplicity assume we have only two predictors. We have then following model + $$ -\log{ \frac{p(\boldsymbol{\beta}\boldsymbol{x})}{1-p(\boldsymbol{\beta}\boldsymbol{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p. +\log{\frac{p(C=1\vert x)}{p(K\vert x)}} = \beta_{10}+\beta_{11}x_1, $$ -Here we defined \( \boldsymbol{x}=[1,x_1,x_2,\dots,x_p] \) and \( \boldsymbol{\beta}=[\beta_0, \beta_1, \dots, \beta_p] \) leading to +and $$ -p(\boldsymbol{\beta}\boldsymbol{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}. +\log{\frac{p(C=2\vert x)}{p(K\vert x)}} = \beta_{20}+\beta_{21}x_1, $$ +and so on till the class \( C=K-1 \) class +$$ +\log{\frac{p(C=K-1\vert x)}{p(K\vert x)}} = \beta_{(K-1)0}+\beta_{(K-1)1}x_1, +$$ + +

    +and the model is specified in term of \( K-1 \) so-called log-odds or +logit transformations. +

    @@ -462,7 +469,7 @@ $$

  • 51
  • 52
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/week38-bs.html b/doc/pub/week38/html/week38-bs.html index 406c60fa3..d5042016a 100644 --- a/doc/pub/week38/html/week38-bs.html +++ b/doc/pub/week38/html/week38-bs.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
  • Further Manipulations
  • Wrapping it up
  • Linear Regression code, Intercept handling first
  • -
  • Another straight Line, the Intercept and Scaling again
  • -
  • Code Examples
  • -
  • Taking out the mean
  • -
  • More complicated Example: The Ising model
  • -
  • Reformulating the problem to suit regression
  • -
  • Linear regression
  • -
  • Singular Value decomposition
  • -
  • The one-dimensional Ising model
  • -
  • Ridge regression
  • -
  • LASSO regression
  • -
  • Performance as function of the regularization parameter
  • -
  • Finding the optimal value of \( \lambda \)
  • -
  • Logistic Regression
  • -
  • Classification problems
  • -
  • Optimization and Deep learning
  • -
  • Basics
  • -
  • Linear classifier
  • -
  • Some selected properties
  • -
  • Simple example
  • -
  • Plotting the mean value for each group
  • -
  • The logistic function
  • -
  • Examples of likelihood functions used in logistic regression and nueral networks
  • -
  • Two parameters
  • -
  • Maximum likelihood
  • -
  • The cost function rewritten
  • -
  • Minimizing the cross entropy
  • -
  • A more compact expression
  • -
  • Extending to more predictors
  • -
  • Including more classes
  • -
  • More classes
  • -
  • Friday September 24
  • -
  • Wisconsin Cancer Data
  • -
  • Using the correlation matrix
  • -
  • Discussing the correlation data
  • -
  • Other measures in classification studies: Cancer Data again
  • -
  • Optimization, the central part of any Machine Learning algortithm
  • -
  • Revisiting our Logistic Regression case
  • -
  • The equations to solve
  • -
  • Solving using Newton-Raphson's method
  • -
  • Brief reminder on Newton-Raphson's method
  • -
  • The equations
  • -
  • Simple geometric interpretation
  • -
  • Extending to more than one variable
  • -
  • Steepest descent
  • -
  • More on Steepest descent
  • -
  • The ideal
  • -
  • The sensitiveness of the gradient descent
  • -
  • Convex functions
  • -
  • Convex function
  • -
  • Conditions on convex functions
  • -
  • More on convex functions
  • -
  • Some simple problems
  • -
  • Friday September 25
  • -
  • Standard steepest descent
  • -
  • Gradient method
  • -
  • Steepest descent method
  • -
  • Steepest descent method
  • -
  • Final expressions
  • -
  • Steepest descent example
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method and iterations
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Conjugate gradient method
  • -
  • Revisiting some of our first Linear Regression Encounters
  • -
  • Gradient descent example
  • -
  • The derivative of the cost/loss function
  • -
  • The Hessian matrix
  • -
  • Simple program
  • -
  • Gradient Descent Example
  • -
  • And a corresponding example using scikit-learn
  • -
  • Gradient descent and Ridge
  • -
  • Program example for gradient descent with Ridge Regression
  • -
  • Using gradient descent methods, limitations
  • +
  • Code Examples
  • +
  • Taking out the mean
  • +
  • More complicated Example: The Ising model
  • +
  • Reformulating the problem to suit regression
  • +
  • Linear regression
  • +
  • Singular Value decomposition
  • +
  • The one-dimensional Ising model
  • +
  • Ridge regression
  • +
  • LASSO regression
  • +
  • Performance as function of the regularization parameter
  • +
  • Finding the optimal value of \( \lambda \)
  • +
  • Logistic Regression
  • +
  • Classification problems
  • +
  • Optimization and Deep learning
  • +
  • Basics
  • +
  • Linear classifier
  • +
  • Some selected properties
  • +
  • Simple example
  • +
  • Plotting the mean value for each group
  • +
  • The logistic function
  • +
  • Examples of likelihood functions used in logistic regression and nueral networks
  • +
  • Two parameters
  • +
  • Maximum likelihood
  • +
  • The cost function rewritten
  • +
  • Minimizing the cross entropy
  • +
  • A more compact expression
  • +
  • Extending to more predictors
  • +
  • Including more classes
  • +
  • More classes
  • +
  • Friday September 24
  • +
  • Wisconsin Cancer Data
  • +
  • Using the correlation matrix
  • +
  • Discussing the correlation data
  • +
  • Other measures in classification studies: Cancer Data again
  • +
  • Optimization, the central part of any Machine Learning algortithm
  • +
  • Revisiting our Logistic Regression case
  • +
  • The equations to solve
  • +
  • Solving using Newton-Raphson's method
  • +
  • Brief reminder on Newton-Raphson's method
  • +
  • The equations
  • +
  • Simple geometric interpretation
  • +
  • Extending to more than one variable
  • +
  • Steepest descent
  • +
  • More on Steepest descent
  • +
  • The ideal
  • +
  • The sensitiveness of the gradient descent
  • +
  • Convex functions
  • +
  • Convex function
  • +
  • Conditions on convex functions
  • +
  • More on convex functions
  • +
  • Some simple problems
  • +
  • Friday September 25
  • +
  • Standard steepest descent
  • +
  • Gradient method
  • +
  • Steepest descent method
  • +
  • Steepest descent method
  • +
  • Final expressions
  • +
  • Steepest descent example
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method and iterations
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Conjugate gradient method
  • +
  • Revisiting some of our first Linear Regression Encounters
  • +
  • Gradient descent example
  • +
  • The derivative of the cost/loss function
  • +
  • The Hessian matrix
  • +
  • Simple program
  • +
  • Gradient Descent Example
  • +
  • And a corresponding example using scikit-learn
  • +
  • Gradient descent and Ridge
  • +
  • Program example for gradient descent with Ridge Regression
  • +
  • Using gradient descent methods, limitations
  • @@ -466,7 +461,7 @@ MathJax.Hub.Config({
  • 9
  • 10
  • ...
  • -
  • 92
  • +
  • 91
  • »
  • diff --git a/doc/pub/week38/html/week38-reveal.html b/doc/pub/week38/html/week38-reveal.html index 758544922..da9264ad6 100644 --- a/doc/pub/week38/html/week38-reveal.html +++ b/doc/pub/week38/html/week38-reveal.html @@ -666,7 +666,7 @@ $$

     

    -What does this mean for thus? And why do we insist on all this? Let us look at some examples. +What does this mean? And why do we insist on all this? Let us look at some examples. @@ -687,6 +687,10 @@ Note also that we do not split the data into training and test. np.random.seed(2021) +def MSE(y_data,y_model): + n = np.size(y_model) + return np.sum((y_data-y_model)**2)/n + def fit_beta(X, y): return np.linalg.pinv(X.T @ X) @ X.T @ y @@ -714,6 +718,12 @@ clf = LinearRegression(fit_intercept=print(f"True beta: {true_beta}") print(f"Fitted beta: {beta}") print(f"Sklearn fitted beta: {clf.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with intercept column: {intercept}") +print(MSE(y,ypredictOwn) +print(f"MSE with intercept column from SKL: {intercept}") +print(MSE(y,ypredictSKL) plt.figure() @@ -742,6 +752,12 @@ intercept = np.mean(y_offset - X_offset @ beta) print(f"Fitted beta (wiothout intercept): {beta}") print(f"Sklearn intercept: {clf.intercept_}") print(f"Sklearn fitted beta (without intercept): {clf.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with Manual intercept: {intercept}") +print(MSE(y,ypredictOwn) +print(f"MSE with Sklearn intercept: {clf.intercept_}") +print(MSE(y,ypredictSKL) plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)") plt.plot(x, clf.predict(X), "--", label="Sklearn (fit_intercept=True)") @@ -750,65 +766,10 @@ plt.legend() plt.show()

    - - - -
    -

    Another straight Line, the Intercept and Scaling again

    -

    -As we said above, the intercept is the value of our output/target variable -when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). -Let us now study a case where we again fit a straight line including the intercept in the design matrix -using ordinary least squares. We are only interested in the MSE value here. +The intercept is the value of our output/target variable +when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). -

    - - -

    import numpy as np
    -import pandas as pd
    -import matplotlib.pyplot as plt
    -from sklearn.model_selection import train_test_split
    -from sklearn import linear_model
    -
    -def MSE(y_data,y_model):
    -    n = np.size(y_model)
    -    return np.sum((y_data-y_model)**2)/n
    -
    -def fit_beta(X, y):
    -    return np.linalg.pinv(X.T @ X) @ X.T @ y
    -
    -
    -
    -# A seed just to ensure that the random numbers are the same for every run.
    -# Useful for eventual debugging.
    -np.random.seed(3155)
    -
    -n = 100
    -x = np.random.rand(n)
    -y = 10.0+5*x
    -
    -
    -Maxpolydegree = 2
    -X = np.zeros((n,Maxpolydegree))
    -
    -for degree in range(Maxpolydegree):
    -    X[:,degree] = x**degree
    -
    -
    -# We split the data in test and training data
    -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    -
    -# Intercept is included in the design matrix
    -# own code first
    -OwnBeta = fit_beta(X_train, y_train)
    -# Scikit-Learn
    -skl = LinearRegression(fit_intercept=False).fit(X_train, y_train)
    -ypredictOwn = X_test @ OwnBeta
    -ypredictSKL = skl.predict(X_test)
    -print(MSE(y_test,ypredictOwn)
    -print(MSE(y_test,ypredictSKL)
    -

    Printing the MSE, we see first that both methods give the same MSE. However, changing the value of the intercept gives a larger or diff --git a/doc/pub/week38/html/week38-solarized.html b/doc/pub/week38/html/week38-solarized.html index 38dcb81ba..c0fabbfa9 100644 --- a/doc/pub/week38/html/week38-solarized.html +++ b/doc/pub/week38/html/week38-solarized.html @@ -101,10 +101,6 @@ div { text-align: justify; text-justify: inter-word; } 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -804,7 +800,7 @@ $$ $$

    -What does this mean for thus? And why do we insist on all this? Let us look at some examples. +What does this mean? And why do we insist on all this? Let us look at some examples.











    @@ -825,6 +821,10 @@ Note also that we do not split the data into training and test. np.random.seed(2021) +def MSE(y_data,y_model): + n = np.size(y_model) + return np.sum((y_data-y_model)**2)/n + def fit_beta(X, y): return np.linalg.pinv(X.T @ X) @ X.T @ y @@ -852,6 +852,12 @@ clf = LinearRegression(fit_intercept=print(f"True beta: {true_beta}") print(f"Fitted beta: {beta}") print(f"Sklearn fitted beta: {clf.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with intercept column: {intercept}") +print(MSE(y,ypredictOwn) +print(f"MSE with intercept column from SKL: {intercept}") +print(MSE(y,ypredictSKL) plt.figure() @@ -880,6 +886,12 @@ intercept = np.mean(y_offset - X_offset @ beta) print(f"Fitted beta (wiothout intercept): {beta}") print(f"Sklearn intercept: {clf.intercept_}") print(f"Sklearn fitted beta (without intercept): {clf.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with Manual intercept: {intercept}") +print(MSE(y,ypredictOwn) +print(f"MSE with Sklearn intercept: {clf.intercept_}") +print(MSE(y,ypredictSKL) plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)") plt.plot(x, clf.predict(X), "--", label="Sklearn (fit_intercept=True)") @@ -889,63 +901,9 @@ plt.legend() plt.show()

    -









    +The intercept is the value of our output/target variable +when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). -

    Another straight Line, the Intercept and Scaling again

    - -

    -As we said above, the intercept is the value of our output/target variable -when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). -Let us now study a case where we again fit a straight line including the intercept in the design matrix -using ordinary least squares. We are only interested in the MSE value here. - -

    - - -

    import numpy as np
    -import pandas as pd
    -import matplotlib.pyplot as plt
    -from sklearn.model_selection import train_test_split
    -from sklearn import linear_model
    -
    -def MSE(y_data,y_model):
    -    n = np.size(y_model)
    -    return np.sum((y_data-y_model)**2)/n
    -
    -def fit_beta(X, y):
    -    return np.linalg.pinv(X.T @ X) @ X.T @ y
    -
    -
    -
    -# A seed just to ensure that the random numbers are the same for every run.
    -# Useful for eventual debugging.
    -np.random.seed(3155)
    -
    -n = 100
    -x = np.random.rand(n)
    -y = 10.0+5*x
    -
    -
    -Maxpolydegree = 2
    -X = np.zeros((n,Maxpolydegree))
    -
    -for degree in range(Maxpolydegree):
    -    X[:,degree] = x**degree
    -
    -
    -# We split the data in test and training data
    -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    -
    -# Intercept is included in the design matrix
    -# own code first
    -OwnBeta = fit_beta(X_train, y_train)
    -# Scikit-Learn
    -skl = LinearRegression(fit_intercept=False).fit(X_train, y_train)
    -ypredictOwn = X_test @ OwnBeta
    -ypredictSKL = skl.predict(X_test)
    -print(MSE(y_test,ypredictOwn)
    -print(MSE(y_test,ypredictSKL)
    -

    Printing the MSE, we see first that both methods give the same MSE. However, changing the value of the intercept gives a larger or diff --git a/doc/pub/week38/html/week38.html b/doc/pub/week38/html/week38.html index b66923b1a..15a917992 100644 --- a/doc/pub/week38/html/week38.html +++ b/doc/pub/week38/html/week38.html @@ -106,10 +106,6 @@ div { text-align: justify; text-justify: inter-word; } 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -809,7 +805,7 @@ $$ $$

    -What does this mean for thus? And why do we insist on all this? Let us look at some examples. +What does this mean? And why do we insist on all this? Let us look at some examples.











    @@ -830,6 +826,10 @@ Note also that we do not split the data into training and test. np.random.seed(2021) +def MSE(y_data,y_model): + n = np.size(y_model) + return np.sum((y_data-y_model)**2)/n + def fit_beta(X, y): return np.linalg.pinv(X.T @ X) @ X.T @ y @@ -857,6 +857,12 @@ clf = LinearRegression(fit_interceptprint(f"True beta: {true_beta}") print(f"Fitted beta: {beta}") print(f"Sklearn fitted beta: {clf.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with intercept column: {intercept}") +print(MSE(y,ypredictOwn) +print(f"MSE with intercept column from SKL: {intercept}") +print(MSE(y,ypredictSKL) plt.figure() @@ -885,6 +891,12 @@ intercept = np. print(f"Fitted beta (wiothout intercept): {beta}") print(f"Sklearn intercept: {clf.intercept_}") print(f"Sklearn fitted beta (without intercept): {clf.coef_}") +ypredictOwn = X @ beta +ypredictSKL = skl.predict(X) +print(f"MSE with Manual intercept: {intercept}") +print(MSE(y,ypredictOwn) +print(f"MSE with Sklearn intercept: {clf.intercept_}") +print(MSE(y,ypredictSKL) plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)") plt.plot(x, clf.predict(X), "--", label="Sklearn (fit_intercept=True)") @@ -894,63 +906,9 @@ plt.legend() plt.show()

    -









    +The intercept is the value of our output/target variable +when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). -

    Another straight Line, the Intercept and Scaling again

    - -

    -As we said above, the intercept is the value of our output/target variable -when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). -Let us now study a case where we again fit a straight line including the intercept in the design matrix -using ordinary least squares. We are only interested in the MSE value here. - -

    - - -

    import numpy as np
    -import pandas as pd
    -import matplotlib.pyplot as plt
    -from sklearn.model_selection import train_test_split
    -from sklearn import linear_model
    -
    -def MSE(y_data,y_model):
    -    n = np.size(y_model)
    -    return np.sum((y_data-y_model)**2)/n
    -
    -def fit_beta(X, y):
    -    return np.linalg.pinv(X.T @ X) @ X.T @ y
    -
    -
    -
    -# A seed just to ensure that the random numbers are the same for every run.
    -# Useful for eventual debugging.
    -np.random.seed(3155)
    -
    -n = 100
    -x = np.random.rand(n)
    -y = 10.0+5*x
    -
    -
    -Maxpolydegree = 2
    -X = np.zeros((n,Maxpolydegree))
    -
    -for degree in range(Maxpolydegree):
    -    X[:,degree] = x**degree
    -
    -
    -# We split the data in test and training data
    -X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
    -
    -# Intercept is included in the design matrix
    -# own code first
    -OwnBeta = fit_beta(X_train, y_train)
    -# Scikit-Learn
    -skl = LinearRegression(fit_intercept=False).fit(X_train, y_train)
    -ypredictOwn = X_test @ OwnBeta
    -ypredictSKL = skl.predict(X_test)
    -print(MSE(y_test,ypredictOwn)
    -print(MSE(y_test,ypredictSKL)
    -

    Printing the MSE, we see first that both methods give the same MSE. However, changing the value of the intercept gives a larger or diff --git a/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz b/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz index 0d38a1c71460e8196452d8c4a122828a23ed364d..822c4fc4f1629aff7b51ffe6b26d72f907d68f94 100644 GIT binary patch literal 193 zcmV;y06za8iwFSBA538Y1MSaC3c@fD2H>uHia9|^nm*hLcHu%0@d7EG+E|;^Bt?6B z`v6@jZi)!`Hb27*!^|ODZ+2N=@77xkAtZ?+7&A@cDM>ij6G~&C5e*rkX-Xp?l*J+Q zfGl^?OJ^+C!zoR5MrlyKn;XW;^246_6?o>KI99^IcHi4dNs!87u2c;-#G0)F(e^Tj vLZKO3pz+!Xjlg9OyeNbfO7e@}YIV}QF@gWgptv73$qwKUS6>_Drw9GylY)k`^|*&Q)50(oW=BpW!CNnJy#QbSj5J u3p=#Hh-)j20IoXVMIoKkieJLU=%eAajly3)<9VLveeD5){_2