diff --git a/doc/pub/week38/html/._week38-bs000.html b/doc/pub/week38/html/._week38-bs000.html index 79aaab611..f40e5396c 100644 --- a/doc/pub/week38/html/._week38-bs000.html +++ b/doc/pub/week38/html/._week38-bs000.html @@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source 2, None, 'code-example-for-cross-validation-and-k-fold-cross-validation'), + ('To think about', 2, None, 'to-think-about'), ('More complicated Example: The Ising model', 2, None, @@ -186,37 +187,38 @@ MathJax.Hub.Config({
-The one-dimensional Ising model with nearest neighbor interaction, no -external field and a constant coupling constant \( J \) is given by - -$$ -\begin{align} - H = -J \sum_{k}^L s_k s_{k + 1}, -\tag{1} -\end{align} -$$ - -
-where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins -in the system is determined by \( L \). For the one-dimensional system -there is no phase transition. - -
-We will look at a system of \( L = 40 \) spins with a coupling constant of -\( J = 1 \). To get enough training data we will generate 10000 states -with their respective energies. +When you are comparing your own code with for example Scikit-Learn's library, there are some minor things to keep in mind. +The example here shows how one can leave out or keep the intercept.
import numpy as np
+import pandas as pd
import matplotlib.pyplot as plt
-from mpl_toolkits.axes_grid1 import make_axes_locatable
-import seaborn as sns
-import scipy.linalg as scl
from sklearn.model_selection import train_test_split
-import tqdm
-sns.set(color_codes=True)
-cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
+from sklearn import linear_model
-L = 40
-n = int(1e4)
+def R2(y_data, y_model):
+ return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
-spins = np.random.choice([-1, 1], size=(n, L))
-J = 1.0
-energies = np.zeros(n)
+# A seed just to ensure that the random numbers are the same for every run.
+# Useful for eventual debugging.
+np.random.seed(3155)
-for i in range(n):
- energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
+n = 100
+x = np.random.rand(n)
+y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
+
+Maxpolydegree = 20
+X = np.zeros((n,Maxpolydegree))
+X[:,0] = 1.0
+
+for polydegree in range(1, Maxpolydegree):
+ for degree in range(polydegree):
+ X[:,degree] = x**degree
+
+
+# We split the data in test and training data
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
+
+# matrix inversion to find beta
+OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train
+print(OLSbeta)
+# and then make the prediction
+ytildeOLS = X_train @ OLSbeta
+print("Training MSE for OLS")
+print(MSE(y_train,ytildeOLS))
+ypredictOLS = X_test @ OLSbeta
+print("Test MSE OLS")
+print(MSE(y_test,ypredictOLS))
+
+p = len(OLSbeta)
+I = np.eye(p,p)
+# Decide which values of lambda to use
+nlambdas = 4
+MSEOwnRidgePredict = np.zeros(nlambdas)
+MSEOwnRidgeTrain = np.zeros(nlambdas)
+MSERidgePredict = np.zeros(nlambdas)
+MSERidgeTrain = np.zeros(nlambdas)
+
+lambdas = np.logspace(-4, 4, nlambdas)
+for i in range(nlambdas):
+ lmb = lambdas[i]
+ OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
+ # include lasso using Scikit-Learn
+ # Note: we include the intercept
+ RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
+ RegRidge.fit(X_train,y_train)
+ # and then make the prediction
+ ytildeOwnRidge = X_train @ OwnRidgeBeta
+ ypredictOwnRidge = X_test @ OwnRidgeBeta
+ ytildeRidge = RegRidge.predict(X_train)
+ ypredictRidge = RegRidge.predict(X_test)
+ MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
+ MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)
+ MSERidgePredict[i] = MSE(y_test,ypredictRidge)
+ MSERidgeTrain[i] = MSE(y_train,ytildeRidge)
+ print("Beta values for own Ridge implementation")
+ print(OwnRidgeBeta)
+ print("Beta values for Scikit-Learn Ridge implementation")
+ print(RegRidge.coef_)
+# Now plot the results
+plt.figure()
+plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r--', label = 'MSE Ridge Test')
+plt.plot(np.log10(lambdas), MSERidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSERidgePredict, 'r--', label = 'MSE Ridge Test')
+
+plt.xlabel('log10(lambda)')
+plt.ylabel('MSE')
+plt.legend()
+plt.show()
-Here we use ordinary least squares -regression to predict the energy for the nearest neighbor -one-dimensional Ising model on a ring, i.e., the endpoints wrap -around. We will use linear regression to fit a value for -the coupling constant to achieve this. -
@@ -309,7 +352,7 @@ the coupling constant to achieve this.
-A more general form for the one-dimensional Ising model is +The one-dimensional Ising model with nearest neighbor interaction, no +external field and a constant coupling constant \( J \) is given by $$ \begin{align} - H = - \sum_j^L \sum_k^L s_j s_k J_{jk}. -\tag{2} + H = -J \sum_{k}^L s_k s_{k + 1}, +\tag{1} \end{align} $$
-Here we allow for interactions beyond the nearest neighbors and a state dependent -coupling constant. This latter expression can be formulated as -a matrix-product -$$ -\begin{align} - \boldsymbol{H} = \boldsymbol{X} J, -\tag{3} -\end{align} -$$ +where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins +in the system is determined by \( L \). For the one-dimensional system +there is no phase transition.
-where \( X_{jk} = s_j s_k \) and \( J \) is a matrix which consists of the -elements \( -J_{jk} \). This form of writing the energy fits perfectly -with the form utilized in linear regression, that is - -$$ -\begin{align} - \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}, -\tag{4} -\end{align} -$$ - -
-We split the data in training and test data as discussed in the previous example +We will look at a system of \( L = 40 \) spins with a coupling constant of +\( J = 1 \). To get enough training data we will generate 10000 states +with their respective energies.
-
X = np.zeros((n, L ** 2))
+import numpy as np
+import matplotlib.pyplot as plt
+from mpl_toolkits.axes_grid1 import make_axes_locatable
+import seaborn as sns
+import scipy.linalg as scl
+from sklearn.model_selection import train_test_split
+import tqdm
+sns.set(color_codes=True)
+cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
+
+L = 40
+n = int(1e4)
+
+spins = np.random.choice([-1, 1], size=(n, L))
+J = 1.0
+
+energies = np.zeros(n)
+
for i in range(n):
- X[i] = np.outer(spins[i], spins[i]).ravel()
-y = energies
-X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
+ energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
+
+Here we use ordinary least squares
+regression to predict the energy for the nearest neighbor
+one-dimensional Ising model on a ring, i.e., the endpoints wrap
+around. We will use linear regression to fit a value for
+the coupling constant to achieve this.
+
@@ -303,7 +312,7 @@ X_train, X_test, y_train, y_test = train_tes
-In the ordinary least squares method we choose the cost function +A more general form for the one-dimensional Ising model is $$ \begin{align} - C(\boldsymbol{X}, \boldsymbol{\beta})= \frac{1}{n}\left\{(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})\right\}. -\tag{5} + H = - \sum_j^L \sum_k^L s_j s_k J_{jk}. +\tag{2} \end{align} $$
-We then find the extremal point of \( C \) by taking the derivative with respect to \( \boldsymbol{\beta} \) as discussed above. -This yields the expression for \( \boldsymbol{\beta} \) to be - +Here we allow for interactions beyond the nearest neighbors and a state dependent +coupling constant. This latter expression can be formulated as +a matrix-product $$ - \boldsymbol{\beta} = \frac{\boldsymbol{X}^T \boldsymbol{y}}{\boldsymbol{X}^T \boldsymbol{X}}, +\begin{align} + \boldsymbol{H} = \boldsymbol{X} J, +\tag{3} +\end{align} $$
-which immediately imposes some requirements on \( \boldsymbol{X} \) as there must exist -an inverse of \( \boldsymbol{X}^T \boldsymbol{X} \). If the expression we are modeling contains an -intercept, i.e., a constant term, we must make sure that the -first column of \( \boldsymbol{X} \) consists of \( 1 \). We do this here +where \( X_{jk} = s_j s_k \) and \( J \) is a matrix which consists of the +elements \( -J_{jk} \). This form of writing the energy fits perfectly +with the form utilized in linear regression, that is + +$$ +\begin{align} + \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}, +\tag{4} +\end{align} +$$ + +
+We split the data in training and test data as discussed in the previous example
-
X_train_own = np.concatenate(
- (np.ones(len(X_train))[:, np.newaxis], X_train),
- axis=1
-)
-X_test_own = np.concatenate(
- (np.ones(len(X_test))[:, np.newaxis], X_test),
- axis=1
-)
-- - -
def ols_inv(x: np.ndarray, y: np.ndarray) -> np.ndarray:
- return scl.inv(x.T @ x) @ (x.T @ y)
-beta = ols_inv(X_train_own, y_train)
+X = np.zeros((n, L ** 2))
+for i in range(n):
+ X[i] = np.outer(spins[i], spins[i]).ravel()
+y = energies
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
@@ -302,7 +306,7 @@ beta = ols_inv(X_train_own, y_train)
-Doing the inversion directly turns out to be a bad idea since the matrix -\( \boldsymbol{X}^T\boldsymbol{X} \) is singular. An alternative approach is to use the singular -value decomposition. Using the definition of the Moore-Penrose -pseudoinverse we can write the equation for \( \boldsymbol{\beta} \) as +In the ordinary least squares method we choose the cost function -$$ - \boldsymbol{\beta} = \boldsymbol{X}^{+}\boldsymbol{y}, -$$ - -
-where the pseudoinverse of \( \boldsymbol{X} \) is given by - -$$ - \boldsymbol{X}^{+} = \frac{\boldsymbol{X}^T}{\boldsymbol{X}^T\boldsymbol{X}}. -$$ - -
-Using singular value decomposition we can decompose the matrix \( \boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma} \boldsymbol{V}^T \), -where \( \boldsymbol{U} \) and \( \boldsymbol{V} \) are orthogonal(unitary) matrices and \( \boldsymbol{\Sigma} \) contains the singular values (more details below). -where \( X^{+} = V\Sigma^{+} U^T \). This reduces the equation for -\( \omega \) to $$ \begin{align} - \boldsymbol{\beta} = \boldsymbol{V}\boldsymbol{\Sigma}^{+} \boldsymbol{U}^T \boldsymbol{y}. -\tag{6} + C(\boldsymbol{X}, \boldsymbol{\beta})= \frac{1}{n}\left\{(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})\right\}. +\tag{5} \end{align} $$
-Note that solving this equation by actually doing the pseudoinverse -(which is what we will do) is not a good idea as this operation scales -as \( \mathcal{O}(n^3) \), where \( n \) is the number of elements in a -general matrix. Instead, doing \( QR \)-factorization and solving the -linear system as an equation would reduce this down to -\( \mathcal{O}(n^2) \) operations. +We then find the extremal point of \( C \) by taking the derivative with respect to \( \boldsymbol{\beta} \) as discussed above. +This yields the expression for \( \boldsymbol{\beta} \) to be + +$$ + \boldsymbol{\beta} = \frac{\boldsymbol{X}^T \boldsymbol{y}}{\boldsymbol{X}^T \boldsymbol{X}}, +$$ + +
+which immediately imposes some requirements on \( \boldsymbol{X} \) as there must exist +an inverse of \( \boldsymbol{X}^T \boldsymbol{X} \). If the expression we are modeling contains an +intercept, i.e., a constant term, we must make sure that the +first column of \( \boldsymbol{X} \) consists of \( 1 \). We do this here
-
def ols_svd(x: np.ndarray, y: np.ndarray) -> np.ndarray:
- u, s, v = scl.svd(x)
- return v.T @ scl.pinv(scl.diagsvd(s, u.shape[0], v.shape[0])) @ u.T @ y
+X_train_own = np.concatenate(
+ (np.ones(len(X_train))[:, np.newaxis], X_train),
+ axis=1
+)
+X_test_own = np.concatenate(
+ (np.ones(len(X_test))[:, np.newaxis], X_test),
+ axis=1
+)
-
beta = ols_svd(X_train_own,y_train)
+def ols_inv(x: np.ndarray, y: np.ndarray) -> np.ndarray:
+ return scl.inv(x.T @ x) @ (x.T @ y)
+beta = ols_inv(X_train_own, y_train)
-
-When extracting the \( J \)-matrix we need to make sure that we remove the intercept, as is done here
-
-
-
-
-
J = beta[1:].reshape(L, L)
-
-
-A way of looking at the coefficients in \( J \) is to plot the matrices as images.
-
-
-
-
-
fig = plt.figure(figsize=(20, 14))
-im = plt.imshow(J, **cmap_args)
-plt.title("OLS", fontsize=18)
-plt.xticks(fontsize=18)
-plt.yticks(fontsize=18)
-cb = fig.colorbar(im)
-cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
-plt.show()
-
-
-It is interesting to note that OLS
-considers both \( J_{j, j + 1} = -0.5 \) and \( J_{j, j - 1} = -0.5 \) as
-valid matrix elements for \( J \).
-In our discussion below on hyperparameters and Ridge and Lasso regression we will see that
-this problem can be removed, partly and only with Lasso regression.
-
-
-In this case our matrix inversion was actually possible. The obvious question now is what is the mathematics behind the SVD?
-
@@ -342,7 +305,7 @@ In this case our matrix inversion was actually possible. The obvious question no
19
20
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs011.html b/doc/pub/week38/html/._week38-bs011.html
index 5aa779321..62f9e0be1 100644
--- a/doc/pub/week38/html/._week38-bs011.html
+++ b/doc/pub/week38/html/._week38-bs011.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,128 +234,74 @@ MathJax.Hub.Config({
-The one-dimensional Ising model
+Singular Value decomposition
-Let us bring back the Ising model again, but now with an additional
-focus on Ridge and Lasso regression as well. We repeat some of the
-basic parts of the Ising model and the setup of the training and test
-data. The one-dimensional Ising model with nearest neighbor
-interaction, no external field and a constant coupling constant \( J \) is
-given by
+Doing the inversion directly turns out to be a bad idea since the matrix
+\( \boldsymbol{X}^T\boldsymbol{X} \) is singular. An alternative approach is to use the singular
+value decomposition. Using the definition of the Moore-Penrose
+pseudoinverse we can write the equation for \( \boldsymbol{\beta} \) as
+$$
+ \boldsymbol{\beta} = \boldsymbol{X}^{+}\boldsymbol{y},
+$$
+
+
+where the pseudoinverse of \( \boldsymbol{X} \) is given by
+
+$$
+ \boldsymbol{X}^{+} = \frac{\boldsymbol{X}^T}{\boldsymbol{X}^T\boldsymbol{X}}.
+$$
+
+
+Using singular value decomposition we can decompose the matrix \( \boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma} \boldsymbol{V}^T \),
+where \( \boldsymbol{U} \) and \( \boldsymbol{V} \) are orthogonal(unitary) matrices and \( \boldsymbol{\Sigma} \) contains the singular values (more details below).
+where \( X^{+} = V\Sigma^{+} U^T \). This reduces the equation for
+\( \omega \) to
$$
\begin{align}
- H = -J \sum_{k}^L s_k s_{k + 1},
-\tag{7}
-\end{align}
-$$
-
-where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins in the system is determined by \( L \). For the one-dimensional system there is no phase transition.
-
-
-We will look at a system of \( L = 40 \) spins with a coupling constant of \( J = 1 \). To get enough training data we will generate 10000 states with their respective energies.
-
-
-
-
-
import numpy as np
-import matplotlib.pyplot as plt
-from mpl_toolkits.axes_grid1 import make_axes_locatable
-import seaborn as sns
-import scipy.linalg as scl
-from sklearn.model_selection import train_test_split
-import sklearn.linear_model as skl
-import tqdm
-sns.set(color_codes=True)
-cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
-
-L = 40
-n = int(1e4)
-
-spins = np.random.choice([-1, 1], size=(n, L))
-J = 1.0
-
-energies = np.zeros(n)
-
-for i in range(n):
- energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
-
-
-A more general form for the one-dimensional Ising model is
-
-$$
-\begin{align}
- H = - \sum_j^L \sum_k^L s_j s_k J_{jk}.
-\tag{8}
+ \boldsymbol{\beta} = \boldsymbol{V}\boldsymbol{\Sigma}^{+} \boldsymbol{U}^T \boldsymbol{y}.
+\tag{6}
\end{align}
$$
-Here we allow for interactions beyond the nearest neighbors and a more
-adaptive coupling matrix. This latter expression can be formulated as
-a matrix-product on the form
-$$
-\begin{align}
- H = X J,
-\tag{9}
-\end{align}
-$$
-
-
-where \( X_{jk} = s_j s_k \) and \( J \) is the matrix consisting of the
-elements \( -J_{jk} \). This form of writing the energy fits perfectly
-with the form utilized in linear regression, viz.
-$$
-\begin{align}
- \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}.
-\tag{10}
-\end{align}
-$$
-
-We organize the data as we did above
-
-
-
-
X = np.zeros((n, L ** 2))
-for i in range(n):
- X[i] = np.outer(spins[i], spins[i]).ravel()
-y = energies
-X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.96)
-
-X_train_own = np.concatenate(
- (np.ones(len(X_train))[:, np.newaxis], X_train),
- axis=1
-)
-
-X_test_own = np.concatenate(
- (np.ones(len(X_test))[:, np.newaxis], X_test),
- axis=1
-)
-
-
-We will do all fitting with Scikit-Learn,
+Note that solving this equation by actually doing the pseudoinverse
+(which is what we will do) is not a good idea as this operation scales
+as \( \mathcal{O}(n^3) \), where \( n \) is the number of elements in a
+general matrix. Instead, doing \( QR \)-factorization and solving the
+linear system as an equation would reduce this down to
+\( \mathcal{O}(n^2) \) operations.
-
clf = skl.LinearRegression().fit(X_train, y_train)
+def ols_svd(x: np.ndarray, y: np.ndarray) -> np.ndarray:
+ u, s, v = scl.svd(x)
+ return v.T @ scl.pinv(scl.diagsvd(s, u.shape[0], v.shape[0])) @ u.T @ y
-When extracting the \( J \)-matrix we make sure to remove the intercept
-
-
J_sk = clf.coef_.reshape(L, L)
+beta = ols_svd(X_train_own,y_train)
-And then we plot the results
+When extracting the \( J \)-matrix we need to make sure that we remove the intercept, as is done here
+
+
+
+
+
J = beta[1:].reshape(L, L)
+
+
+A way of looking at the coefficients in \( J \) is to plot the matrices as images.
+
fig = plt.figure(figsize=(20, 14))
-im = plt.imshow(J_sk, **cmap_args)
-plt.title("LinearRegression from Scikit-learn", fontsize=18)
+im = plt.imshow(J, **cmap_args)
+plt.title("OLS", fontsize=18)
plt.xticks(fontsize=18)
plt.yticks(fontsize=18)
cb = fig.colorbar(im)
@@ -361,7 +309,14 @@ cb.ax.se
plt.show()
-The results perfectly with our previous discussion where we used our own code.
+It is interesting to note that OLS
+considers both \( J_{j, j + 1} = -0.5 \) and \( J_{j, j - 1} = -0.5 \) as
+valid matrix elements for \( J \).
+In our discussion below on hyperparameters and Ridge and Lasso regression we will see that
+this problem can be removed, partly and only with Lasso regression.
+
+
+In this case our matrix inversion was actually possible. The obvious question now is what is the mathematics behind the SVD?
@@ -389,7 +344,7 @@ The results perfectly with our previous discussion where we used our own code.
20
21
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs012.html b/doc/pub/week38/html/._week38-bs012.html
index c64e2194e..1c0c3f55c 100644
--- a/doc/pub/week38/html/._week38-bs012.html
+++ b/doc/pub/week38/html/._week38-bs012.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,38 +234,137 @@ MathJax.Hub.Config({
-Ridge regression
+The one-dimensional Ising model
-Having explored the ordinary least squares we move on to ridge
-regression. In ridge regression we include a regularizer. This
-involves a new cost function which leads to a new estimate for the
-weights \( \boldsymbol{\beta} \). This results in a penalized regression problem. The
-cost function is given by
+Let us bring back the Ising model again, but now with an additional
+focus on Ridge and Lasso regression as well. We repeat some of the
+basic parts of the Ising model and the setup of the training and test
+data. The one-dimensional Ising model with nearest neighbor
+interaction, no external field and a constant coupling constant \( J \) is
+given by
$$
\begin{align}
- C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta}.
-\tag{11}
+ H = -J \sum_{k}^L s_k s_{k + 1},
+\tag{7}
\end{align}
$$
+where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins in the system is determined by \( L \). For the one-dimensional system there is no phase transition.
+
+
+We will look at a system of \( L = 40 \) spins with a coupling constant of \( J = 1 \). To get enough training data we will generate 10000 states with their respective energies.
+
-
_lambda = 0.1
-clf_ridge = skl.Ridge(alpha=_lambda).fit(X_train, y_train)
-J_ridge_sk = clf_ridge.coef_.reshape(L, L)
-fig = plt.figure(figsize=(20, 14))
-im = plt.imshow(J_ridge_sk, **cmap_args)
-plt.title("Ridge from Scikit-learn", fontsize=18)
+import numpy as np
+import matplotlib.pyplot as plt
+from mpl_toolkits.axes_grid1 import make_axes_locatable
+import seaborn as sns
+import scipy.linalg as scl
+from sklearn.model_selection import train_test_split
+import sklearn.linear_model as skl
+import tqdm
+sns.set(color_codes=True)
+cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
+
+L = 40
+n = int(1e4)
+
+spins = np.random.choice([-1, 1], size=(n, L))
+J = 1.0
+
+energies = np.zeros(n)
+
+for i in range(n):
+ energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
+
+
+A more general form for the one-dimensional Ising model is
+
+$$
+\begin{align}
+ H = - \sum_j^L \sum_k^L s_j s_k J_{jk}.
+\tag{8}
+\end{align}
+$$
+
+
+Here we allow for interactions beyond the nearest neighbors and a more
+adaptive coupling matrix. This latter expression can be formulated as
+a matrix-product on the form
+$$
+\begin{align}
+ H = X J,
+\tag{9}
+\end{align}
+$$
+
+
+where \( X_{jk} = s_j s_k \) and \( J \) is the matrix consisting of the
+elements \( -J_{jk} \). This form of writing the energy fits perfectly
+with the form utilized in linear regression, viz.
+$$
+\begin{align}
+ \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}.
+\tag{10}
+\end{align}
+$$
+
+We organize the data as we did above
+
+
+
+
X = np.zeros((n, L ** 2))
+for i in range(n):
+ X[i] = np.outer(spins[i], spins[i]).ravel()
+y = energies
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.96)
+
+X_train_own = np.concatenate(
+ (np.ones(len(X_train))[:, np.newaxis], X_train),
+ axis=1
+)
+
+X_test_own = np.concatenate(
+ (np.ones(len(X_test))[:, np.newaxis], X_test),
+ axis=1
+)
+
+
+We will do all fitting with Scikit-Learn,
+
+
+
+
+
clf = skl.LinearRegression().fit(X_train, y_train)
+
+
+When extracting the \( J \)-matrix we make sure to remove the intercept
+
+
+
+
J_sk = clf.coef_.reshape(L, L)
+
+
+And then we plot the results
+
+
+
+
fig = plt.figure(figsize=(20, 14))
+im = plt.imshow(J_sk, **cmap_args)
+plt.title("LinearRegression from Scikit-learn", fontsize=18)
plt.xticks(fontsize=18)
plt.yticks(fontsize=18)
cb = fig.colorbar(im)
cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
-
plt.show()
+
+The results perfectly with our previous discussion where we used our own code.
+
@@ -290,7 +391,7 @@ plt.show()
21
22
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs013.html b/doc/pub/week38/html/._week38-bs013.html
index 470da7766..7b74c3d07 100644
--- a/doc/pub/week38/html/._week38-bs013.html
+++ b/doc/pub/week38/html/._week38-bs013.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,29 +234,31 @@ MathJax.Hub.Config({
-LASSO regression
+Ridge regression
-In the Least Absolute Shrinkage and Selection Operator (LASSO)-method we get a third cost function.
+Having explored the ordinary least squares we move on to ridge
+regression. In ridge regression we include a regularizer. This
+involves a new cost function which leads to a new estimate for the
+weights \( \boldsymbol{\beta} \). This results in a penalized regression problem. The
+cost function is given by
$$
\begin{align}
- C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \sqrt{\boldsymbol{\beta}^T\boldsymbol{\beta}}.
-\tag{12}
+ C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta}.
+\tag{11}
\end{align}
$$
-
-Finding the extremal point of this cost function is not so straight-forward as in least squares and ridge. We will therefore rely solely on the function ``Lasso`` from Scikit-Learn.
-
-
clf_lasso = skl.Lasso(alpha=_lambda).fit(X_train, y_train)
-J_lasso_sk = clf_lasso.coef_.reshape(L, L)
+_lambda = 0.1
+clf_ridge = skl.Ridge(alpha=_lambda).fit(X_train, y_train)
+J_ridge_sk = clf_ridge.coef_.reshape(L, L)
fig = plt.figure(figsize=(20, 14))
-im = plt.imshow(J_lasso_sk, **cmap_args)
-plt.title("Lasso from Scikit-learn", fontsize=18)
+im = plt.imshow(J_ridge_sk, **cmap_args)
+plt.title("Ridge from Scikit-learn", fontsize=18)
plt.xticks(fontsize=18)
plt.yticks(fontsize=18)
cb = fig.colorbar(im)
@@ -262,11 +266,6 @@ cb.ax.se
plt.show()
-
-It is quite striking how LASSO breaks the symmetry of the coupling
-constant as opposed to ridge and OLS. We get a sparse solution with
-\( J_{j, j + 1} = -1 \).
-
@@ -293,7 +292,7 @@ constant as opposed to ridge and OLS. We get a sparse solution with
22
23
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs014.html b/doc/pub/week38/html/._week38-bs014.html
index 291eb6e27..5ab1015a6 100644
--- a/doc/pub/week38/html/._week38-bs014.html
+++ b/doc/pub/week38/html/._week38-bs014.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,56 +234,40 @@ MathJax.Hub.Config({
-Performance as function of the regularization parameter
+LASSO regression
-We see how the different models perform for a different set of values for \( \lambda \).
+In the Least Absolute Shrinkage and Selection Operator (LASSO)-method we get a third cost function.
+
+$$
+\begin{align}
+ C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \sqrt{\boldsymbol{\beta}^T\boldsymbol{\beta}}.
+\tag{12}
+\end{align}
+$$
+
+
+Finding the extremal point of this cost function is not so straight-forward as in least squares and ridge. We will therefore rely solely on the function ``Lasso`` from Scikit-Learn.
-
lambdas = np.logspace(-4, 5, 10)
-
-train_errors = {
- "ols_sk": np.zeros(lambdas.size),
- "ridge_sk": np.zeros(lambdas.size),
- "lasso_sk": np.zeros(lambdas.size)
-}
-
-test_errors = {
- "ols_sk": np.zeros(lambdas.size),
- "ridge_sk": np.zeros(lambdas.size),
- "lasso_sk": np.zeros(lambdas.size)
-}
-
-plot_counter = 1
-
-fig = plt.figure(figsize=(32, 54))
-
-for i, _lambda in enumerate(tqdm.tqdm(lambdas)):
- for key, method in zip(
- ["ols_sk", "ridge_sk", "lasso_sk"],
- [skl.LinearRegression(), skl.Ridge(alpha=_lambda), skl.Lasso(alpha=_lambda)]
- ):
- method = method.fit(X_train, y_train)
-
- train_errors[key][i] = method.score(X_train, y_train)
- test_errors[key][i] = method.score(X_test, y_test)
-
- omega = method.coef_.reshape(L, L)
-
- plt.subplot(10, 5, plot_counter)
- plt.imshow(omega, **cmap_args)
- plt.title(r"%s, $\lambda = %.4f$" % (key, _lambda))
- plot_counter += 1
+clf_lasso = skl.Lasso(alpha=_lambda).fit(X_train, y_train)
+J_lasso_sk = clf_lasso.coef_.reshape(L, L)
+fig = plt.figure(figsize=(20, 14))
+im = plt.imshow(J_lasso_sk, **cmap_args)
+plt.title("Lasso from Scikit-learn", fontsize=18)
+plt.xticks(fontsize=18)
+plt.yticks(fontsize=18)
+cb = fig.colorbar(im)
+cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
plt.show()
-We see that LASSO reaches a good solution for low
-values of \( \lambda \), but will "wither" when we increase \( \lambda \) too
-much. Ridge is more stable over a larger range of values for
-\( \lambda \), but eventually also fades away.
+It is quite striking how LASSO breaks the symmetry of the coupling
+constant as opposed to ridge and OLS. We get a sparse solution with
+\( J_{j, j + 1} = -1 \).
@@ -309,7 +295,7 @@ much. Ridge is more stable over a larger range of values for
23
24
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs015.html b/doc/pub/week38/html/._week38-bs015.html
index 45b429ee9..9cbaf2ba1 100644
--- a/doc/pub/week38/html/._week38-bs015.html
+++ b/doc/pub/week38/html/._week38-bs015.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,54 +234,56 @@ MathJax.Hub.Config({
-Finding the optimal value of \( \lambda \)
+Performance as function of the regularization parameter
-To determine which value of \( \lambda \) is best we plot the accuracy of
-the models when predicting the training and the testing set. We expect
-the accuracy of the training set to be quite good, but if the accuracy
-of the testing set is much lower this tells us that we might be
-subject to an overfit model. The ideal scenario is an accuracy on the
-testing set that is close to the accuracy of the training set.
+We see how the different models perform for a different set of values for \( \lambda \).
-
fig = plt.figure(figsize=(20, 14))
+lambdas = np.logspace(-4, 5, 10)
-colors = {
- "ols_sk": "r",
- "ridge_sk": "y",
- "lasso_sk": "c"
+train_errors = {
+ "ols_sk": np.zeros(lambdas.size),
+ "ridge_sk": np.zeros(lambdas.size),
+ "lasso_sk": np.zeros(lambdas.size)
}
-for key in train_errors:
- plt.semilogx(
- lambdas,
- train_errors[key],
- colors[key],
- label="Train {0}".format(key),
- linewidth=4.0
- )
+test_errors = {
+ "ols_sk": np.zeros(lambdas.size),
+ "ridge_sk": np.zeros(lambdas.size),
+ "lasso_sk": np.zeros(lambdas.size)
+}
+
+plot_counter = 1
+
+fig = plt.figure(figsize=(32, 54))
+
+for i, _lambda in enumerate(tqdm.tqdm(lambdas)):
+ for key, method in zip(
+ ["ols_sk", "ridge_sk", "lasso_sk"],
+ [skl.LinearRegression(), skl.Ridge(alpha=_lambda), skl.Lasso(alpha=_lambda)]
+ ):
+ method = method.fit(X_train, y_train)
+
+ train_errors[key][i] = method.score(X_train, y_train)
+ test_errors[key][i] = method.score(X_test, y_test)
+
+ omega = method.coef_.reshape(L, L)
+
+ plt.subplot(10, 5, plot_counter)
+ plt.imshow(omega, **cmap_args)
+ plt.title(r"%s, $\lambda = %.4f$" % (key, _lambda))
+ plot_counter += 1
-for key in test_errors:
- plt.semilogx(
- lambdas,
- test_errors[key],
- colors[key] + "--",
- label="Test {0}".format(key),
- linewidth=4.0
- )
-plt.legend(loc="best", fontsize=18)
-plt.xlabel(r"$\lambda$", fontsize=18)
-plt.ylabel(r"$R^2$", fontsize=18)
-plt.tick_params(labelsize=18)
plt.show()
-From the above figure we can see that LASSO with \( \lambda = 10^{-2} \)
-achieves a very good accuracy on the test set. This by far surpasses the
-other models for all values of \( \lambda \).
+We see that LASSO reaches a good solution for low
+values of \( \lambda \), but will "wither" when we increase \( \lambda \) too
+much. Ridge is more stable over a larger range of values for
+\( \lambda \), but eventually also fades away.
@@ -307,7 +311,7 @@ other models for all values of \( \lambda \).
24
25
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs016.html b/doc/pub/week38/html/._week38-bs016.html
index 78adc3bfb..c5ccba0b1 100644
--- a/doc/pub/week38/html/._week38-bs016.html
+++ b/doc/pub/week38/html/._week38-bs016.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -230,22 +232,56 @@ MathJax.Hub.Config({
-
+
-Logistic Regression
+Finding the optimal value of \( \lambda \)
-In linear regression our main interest was centered on learning the
-coefficients of a functional fit (say a polynomial) in order to be
-able to predict the response of a continuous variable on some unseen
-data. The fit to the continuous variable \( y_i \) is based on some
-independent variables \( \hat{x}_i \). Linear regression resulted in
-analytical expressions for standard ordinary Least Squares or Ridge
-regression (in terms of matrices to invert) for several quantities,
-ranging from the variance and thereby the confidence intervals of the
-parameters \( \hat{\beta} \) to the mean squared error. If we can invert
-the product of the design matrices, linear regression gives then a
-simple recipe for fitting our data.
+To determine which value of \( \lambda \) is best we plot the accuracy of
+the models when predicting the training and the testing set. We expect
+the accuracy of the training set to be quite good, but if the accuracy
+of the testing set is much lower this tells us that we might be
+subject to an overfit model. The ideal scenario is an accuracy on the
+testing set that is close to the accuracy of the training set.
+
+
+
+
+
fig = plt.figure(figsize=(20, 14))
+
+colors = {
+ "ols_sk": "r",
+ "ridge_sk": "y",
+ "lasso_sk": "c"
+}
+
+for key in train_errors:
+ plt.semilogx(
+ lambdas,
+ train_errors[key],
+ colors[key],
+ label="Train {0}".format(key),
+ linewidth=4.0
+ )
+
+for key in test_errors:
+ plt.semilogx(
+ lambdas,
+ test_errors[key],
+ colors[key] + "--",
+ label="Test {0}".format(key),
+ linewidth=4.0
+ )
+plt.legend(loc="best", fontsize=18)
+plt.xlabel(r"$\lambda$", fontsize=18)
+plt.ylabel(r"$R^2$", fontsize=18)
+plt.tick_params(labelsize=18)
+plt.show()
+
+
+From the above figure we can see that LASSO with \( \lambda = 10^{-2} \)
+achieves a very good accuracy on the test set. This by far surpasses the
+other models for all values of \( \lambda \).
@@ -273,7 +309,7 @@ simple recipe for fitting our data.
25
26
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs017.html b/doc/pub/week38/html/._week38-bs017.html
index 124957af8..d1e72a95e 100644
--- a/doc/pub/week38/html/._week38-bs017.html
+++ b/doc/pub/week38/html/._week38-bs017.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,25 +234,20 @@ MathJax.Hub.Config({
-Classification problems
+Logistic Regression
-Classification problems, however, are concerned with outcomes taking
-the form of discrete variables (i.e. categories). We may for example,
-on the basis of DNA sequencing for a number of patients, like to find
-out which mutations are important for a certain disease; or based on
-scans of various patients' brains, figure out if there is a tumor or
-not; or given a specific physical system, we'd like to identify its
-state, say whether it is an ordered or disordered system (typical
-situation in solid state physics); or classify the status of a
-patient, whether she/he has a stroke or not and many other similar
-situations.
-
-
-The most common situation we encounter when we apply logistic
-regression is that of two possible outcomes, normally denoted as a
-binary outcome, true or false, positive or negative, success or
-failure etc.
+In linear regression our main interest was centered on learning the
+coefficients of a functional fit (say a polynomial) in order to be
+able to predict the response of a continuous variable on some unseen
+data. The fit to the continuous variable \( y_i \) is based on some
+independent variables \( \hat{x}_i \). Linear regression resulted in
+analytical expressions for standard ordinary Least Squares or Ridge
+regression (in terms of matrices to invert) for several quantities,
+ranging from the variance and thereby the confidence intervals of the
+parameters \( \hat{\beta} \) to the mean squared error. If we can invert
+the product of the design matrices, linear regression gives then a
+simple recipe for fitting our data.
@@ -278,7 +275,7 @@ failure etc.
26
27
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs018.html b/doc/pub/week38/html/._week38-bs018.html
index 8c0ab2382..a677677a6 100644
--- a/doc/pub/week38/html/._week38-bs018.html
+++ b/doc/pub/week38/html/._week38-bs018.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -230,25 +232,27 @@ MathJax.Hub.Config({
-
+
-Optimization and Deep learning
+Classification problems
-Logistic regression will also serve as our stepping stone towards
-neural network algorithms and supervised deep learning. For logistic
-learning, the minimization of the cost function leads to a non-linear
-equation in the parameters \( \hat{\beta} \). The optimization of the
-problem calls therefore for minimization algorithms. This forms the
-bottle neck of all machine learning algorithms, namely how to find
-reliable minima of a multi-variable function. This leads us to the
-family of gradient descent methods. The latter are the working horses
-of basically all modern machine learning algorithms.
+Classification problems, however, are concerned with outcomes taking
+the form of discrete variables (i.e. categories). We may for example,
+on the basis of DNA sequencing for a number of patients, like to find
+out which mutations are important for a certain disease; or based on
+scans of various patients' brains, figure out if there is a tumor or
+not; or given a specific physical system, we'd like to identify its
+state, say whether it is an ordered or disordered system (typical
+situation in solid state physics); or classify the status of a
+patient, whether she/he has a stroke or not and many other similar
+situations.
-We note also that many of the topics discussed here on logistic
-regression are also commonly used in modern supervised Deep Learning
-models, as we will see later.
+The most common situation we encounter when we apply logistic
+regression is that of two possible outcomes, normally denoted as a
+binary outcome, true or false, positive or negative, success or
+failure etc.
@@ -276,7 +280,7 @@ models, as we will see later.
27
28
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs019.html b/doc/pub/week38/html/._week38-bs019.html
index 020ba37d6..a54a8f863 100644
--- a/doc/pub/week38/html/._week38-bs019.html
+++ b/doc/pub/week38/html/._week38-bs019.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -230,31 +232,25 @@ MathJax.Hub.Config({
-
+
-Basics
+Optimization and Deep learning
-We consider the case where the dependent variables, also called the
-responses or the outcomes, \( y_i \) are discrete and only take values
-from \( k=0,\dots,K-1 \) (i.e. \( K \) classes).
+Logistic regression will also serve as our stepping stone towards
+neural network algorithms and supervised deep learning. For logistic
+learning, the minimization of the cost function leads to a non-linear
+equation in the parameters \( \hat{\beta} \). The optimization of the
+problem calls therefore for minimization algorithms. This forms the
+bottle neck of all machine learning algorithms, namely how to find
+reliable minima of a multi-variable function. This leads us to the
+family of gradient descent methods. The latter are the working horses
+of basically all modern machine learning algorithms.
-The goal is to predict the
-output classes from the design matrix \( \hat{X}\in\mathbb{R}^{n\times p} \)
-made of \( n \) samples, each of which carries \( p \) features or predictors. The
-primary goal is to identify the classes to which new unseen samples
-belong.
-
-
-Let us specialize to the case of two classes only, with outputs
-\( y_i=0 \) and \( y_i=1 \). Our outcomes could represent the status of a
-credit card user that could default or not on her/his credit card
-debt. That is
-
-$$
-y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}.
-$$
+We note also that many of the topics discussed here on logistic
+regression are also commonly used in modern supervised Deep Learning
+models, as we will see later.
@@ -282,7 +278,7 @@ $$
28
29
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs020.html b/doc/pub/week38/html/._week38-bs020.html
index 0698e4e45..dccdc57e7 100644
--- a/doc/pub/week38/html/._week38-bs020.html
+++ b/doc/pub/week38/html/._week38-bs020.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -230,28 +232,31 @@ MathJax.Hub.Config({
-
+
-Linear classifier
+Basics
-Before moving to the logistic model, let us try to use our linear
-regression model to classify these two outcomes. We could for example
-fit a linear model to the default case if \( y_i > 0.5 \) and the no
-default case \( y_i \leq 0.5 \).
+We consider the case where the dependent variables, also called the
+responses or the outcomes, \( y_i \) are discrete and only take values
+from \( k=0,\dots,K-1 \) (i.e. \( K \) classes).
-We would then have our
-weighted linear combination, namely
-$$
-\begin{equation}
-\hat{y} = \hat{X}^T\hat{\beta} + \hat{\epsilon},
-\tag{13}
-\end{equation}
-$$
+The goal is to predict the
+output classes from the design matrix \( \hat{X}\in\mathbb{R}^{n\times p} \)
+made of \( n \) samples, each of which carries \( p \) features or predictors. The
+primary goal is to identify the classes to which new unseen samples
+belong.
-where \( \hat{y} \) is a vector representing the possible outcomes, \( \hat{X} \) is our
-\( n\times p \) design matrix and \( \hat{\beta} \) represents our estimators/predictors.
+
+Let us specialize to the case of two classes only, with outputs
+\( y_i=0 \) and \( y_i=1 \). Our outcomes could represent the status of a
+credit card user that could default or not on her/his credit card
+debt. That is
+
+$$
+y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}.
+$$
@@ -279,7 +284,7 @@ where \( \hat{y} \) is a vector representing the possible outcomes, \( \hat{X} \
29
30
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs021.html b/doc/pub/week38/html/._week38-bs021.html
index 3b4bf701d..1d1525021 100644
--- a/doc/pub/week38/html/._week38-bs021.html
+++ b/doc/pub/week38/html/._week38-bs021.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,24 +234,26 @@ MathJax.Hub.Config({
-Some selected properties
+Linear classifier
-The main problem with our function is that it takes values on the
-entire real axis. In the case of logistic regression, however, the
-labels \( y_i \) are discrete variables. A typical example is the credit
-card data discussed below here, where we can set the state of
-defaulting the debt to \( y_i=1 \) and not to \( y_i=0 \) for one the persons
-in the data set (see the full example below).
+Before moving to the logistic model, let us try to use our linear
+regression model to classify these two outcomes. We could for example
+fit a linear model to the default case if \( y_i > 0.5 \) and the no
+default case \( y_i \leq 0.5 \).
-One simple way to get a discrete output is to have sign
-functions that map the output of a linear regressor to values \( \{0,1\} \),
-\( f(s_i)=sign(s_i)=1 \) if \( s_i\ge 0 \) and 0 if otherwise.
-We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron" model in the machine learning
-literature. This model is extremely simple. However, in many cases it is more
-favorable to use a ``soft" classifier that outputs
-the probability of a given category. This leads us to the logistic function.
+We would then have our
+weighted linear combination, namely
+$$
+\begin{equation}
+\hat{y} = \hat{X}^T\hat{\beta} + \hat{\epsilon},
+\tag{13}
+\end{equation}
+$$
+
+where \( \hat{y} \) is a vector representing the possible outcomes, \( \hat{X} \) is our
+\( n\times p \) design matrix and \( \hat{\beta} \) represents our estimators/predictors.
@@ -277,7 +281,7 @@ the probability of a given category. This leads us to the logistic function.
30
31
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs022.html b/doc/pub/week38/html/._week38-bs022.html
index 1eea073d0..6197cf7c8 100644
--- a/doc/pub/week38/html/._week38-bs022.html
+++ b/doc/pub/week38/html/._week38-bs022.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,69 +234,25 @@ MathJax.Hub.Config({
-Simple example
+Some selected properties
-The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful.
+The main problem with our function is that it takes values on the
+entire real axis. In the case of logistic regression, however, the
+labels \( y_i \) are discrete variables. A typical example is the credit
+card data discussed below here, where we can set the state of
+defaulting the debt to \( y_i=1 \) and not to \( y_i=0 \) for one the persons
+in the data set (see the full example below).
+One simple way to get a discrete output is to have sign
+functions that map the output of a linear regressor to values \( \{0,1\} \),
+\( f(s_i)=sign(s_i)=1 \) if \( s_i\ge 0 \) and 0 if otherwise.
+We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron" model in the machine learning
+literature. This model is extremely simple. However, in many cases it is more
+favorable to use a ``soft" classifier that outputs
+the probability of a given category. This leads us to the logistic function.
-
-
# Common imports
-import os
-import numpy as np
-import pandas as pd
-import matplotlib.pyplot as plt
-from sklearn.linear_model import LinearRegression, Ridge, Lasso
-from sklearn.model_selection import train_test_split
-from sklearn.utils import resample
-from sklearn.metrics import mean_squared_error
-from IPython.display import display
-from pylab import plt, mpl
-plt.style.use('seaborn')
-mpl.rcParams['font.family'] = 'serif'
-
-# Where to save the figures and data files
-PROJECT_ROOT_DIR = "Results"
-FIGURE_ID = "Results/FigureFiles"
-DATA_ID = "DataFiles/"
-
-if not os.path.exists(PROJECT_ROOT_DIR):
- os.mkdir(PROJECT_ROOT_DIR)
-
-if not os.path.exists(FIGURE_ID):
- os.makedirs(FIGURE_ID)
-
-if not os.path.exists(DATA_ID):
- os.makedirs(DATA_ID)
-
-def image_path(fig_id):
- return os.path.join(FIGURE_ID, fig_id)
-
-def data_path(dat_id):
- return os.path.join(DATA_ID, dat_id)
-
-def save_fig(fig_id):
- plt.savefig(image_path(fig_id) + ".png", format='png')
-
-infile = open(data_path("chddata.csv"),'r')
-
-# Read the chd data as csv file and organize the data into arrays with age group, age, and chd
-chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))
-chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']
-output = chd['CHD']
-age = chd['Age']
-agegroup = chd['Agegroup']
-numberID = chd['ID']
-display(chd)
-
-plt.scatter(age, output, marker='o')
-plt.axis([18,70.0,-0.1, 1.2])
-plt.xlabel(r'Age')
-plt.ylabel(r'CHD')
-plt.title(r'Age distribution and Coronary heart disease')
-plt.show()
-
@@ -321,7 +279,7 @@ plt.show()
31
32
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs023.html b/doc/pub/week38/html/._week38-bs023.html
index 18aff2651..7523d7429 100644
--- a/doc/pub/week38/html/._week38-bs023.html
+++ b/doc/pub/week38/html/._week38-bs023.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,42 +234,69 @@ MathJax.Hub.Config({
-Plotting the mean value for each group
+Simple example
-What we could attempt however is to plot the mean value for each group.
+The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful.
-
agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])
-group = np.array([1, 2, 3, 4, 5, 6, 7, 8])
-plt.plot(group, agegroupmean, "r-")
-plt.axis([0,9,0, 1.0])
-plt.xlabel(r'Age group')
-plt.ylabel(r'CHD mean values')
-plt.title(r'Mean values for each age group')
+# Common imports
+import os
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.linear_model import LinearRegression, Ridge, Lasso
+from sklearn.model_selection import train_test_split
+from sklearn.utils import resample
+from sklearn.metrics import mean_squared_error
+from IPython.display import display
+from pylab import plt, mpl
+plt.style.use('seaborn')
+mpl.rcParams['font.family'] = 'serif'
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+ os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+ os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+ os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+ return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+ return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+ plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("chddata.csv"),'r')
+
+# Read the chd data as csv file and organize the data into arrays with age group, age, and chd
+chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))
+chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']
+output = chd['CHD']
+age = chd['Age']
+agegroup = chd['Agegroup']
+numberID = chd['ID']
+display(chd)
+
+plt.scatter(age, output, marker='o')
+plt.axis([18,70.0,-0.1, 1.2])
+plt.xlabel(r'Age')
+plt.ylabel(r'CHD')
+plt.title(r'Age distribution and Coronary heart disease')
plt.show()
-
-We are now trying to find a function \( f(y\vert x) \), that is a function which gives us an expected value for the output \( y \) with a given input \( x \).
-In standard linear regression with a linear dependence on \( x \), we would write this in terms of our model
-$$
-f(y_i\vert x_i)=\beta_0+\beta_1 x_i.
-$$
-
-
-This expression implies however that \( f(y_i\vert x_i) \) could take any
-value from minus infinity to plus infinity. If we however let
-\( f(y\vert y) \) be represented by the mean value, the above example
-shows us that we can constrain the function to take values between
-zero and one, that is we have \( 0 \le f(y_i\vert x_i) \le 1 \). Looking
-at our last curve we see also that it has an S-shaped form. This leads
-us to a very popular model for the function \( f \), namely the so-called
-Sigmoid function or logistic model. We will consider this function as
-representing the probability for finding a value of \( y_i \) with a given
-\( x_i \).
-
@@ -294,7 +323,7 @@ representing the probability for finding a value of \( y_i \) with a given
32
33
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs024.html b/doc/pub/week38/html/._week38-bs024.html
index 1f824172b..28a04ceac 100644
--- a/doc/pub/week38/html/._week38-bs024.html
+++ b/doc/pub/week38/html/._week38-bs024.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,25 +234,41 @@ MathJax.Hub.Config({
-The logistic function
+Plotting the mean value for each group
-Another widely studied model, is the so-called
-perceptron model, which is an example of a "hard classification" model. We
-will encounter this model when we discuss neural networks as
-well. Each datapoint is deterministically assigned to a category (i.e
-\( y_i=0 \) or \( y_i=1 \)). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a "soft"
-classifier that outputs the probability of a given category rather
-than a single value. For example, given \( x_i \), the classifier
-outputs the probability of being in a category \( k \). Logistic regression
-is the most common example of a so-called soft classifier. In logistic
-regression, the probability that a data point \( x_i \)
-belongs to a category \( y_i=\{0,1\} \) is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event,
+What we could attempt however is to plot the mean value for each group.
+
+
+
+
+
agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])
+group = np.array([1, 2, 3, 4, 5, 6, 7, 8])
+plt.plot(group, agegroupmean, "r-")
+plt.axis([0,9,0, 1.0])
+plt.xlabel(r'Age group')
+plt.ylabel(r'CHD mean values')
+plt.title(r'Mean values for each age group')
+plt.show()
+
+
+We are now trying to find a function \( f(y\vert x) \), that is a function which gives us an expected value for the output \( y \) with a given input \( x \).
+In standard linear regression with a linear dependence on \( x \), we would write this in terms of our model
$$
-p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}.
+f(y_i\vert x_i)=\beta_0+\beta_1 x_i.
$$
-Note that \( 1-p(t)= p(-t) \).
+
+This expression implies however that \( f(y_i\vert x_i) \) could take any
+value from minus infinity to plus infinity. If we however let
+\( f(y\vert y) \) be represented by the mean value, the above example
+shows us that we can constrain the function to take values between
+zero and one, that is we have \( 0 \le f(y_i\vert x_i) \le 1 \). Looking
+at our last curve we see also that it has an S-shaped form. This leads
+us to a very popular model for the function \( f \), namely the so-called
+Sigmoid function or logistic model. We will consider this function as
+representing the probability for finding a value of \( y_i \) with a given
+\( x_i \).
@@ -278,7 +296,7 @@ Note that \( 1-p(t)= p(-t) \).
33
34
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs025.html b/doc/pub/week38/html/._week38-bs025.html
index 2fad1f375..f1d5ce2fb 100644
--- a/doc/pub/week38/html/._week38-bs025.html
+++ b/doc/pub/week38/html/._week38-bs025.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,69 +234,26 @@ MathJax.Hub.Config({
-Examples of likelihood functions used in logistic regression and nueral networks
+The logistic function
-The following code plots the logistic function, the step function and other functions we will encounter from here and on.
+Another widely studied model, is the so-called
+perceptron model, which is an example of a "hard classification" model. We
+will encounter this model when we discuss neural networks as
+well. Each datapoint is deterministically assigned to a category (i.e
+\( y_i=0 \) or \( y_i=1 \)). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a "soft"
+classifier that outputs the probability of a given category rather
+than a single value. For example, given \( x_i \), the classifier
+outputs the probability of being in a category \( k \). Logistic regression
+is the most common example of a so-called soft classifier. In logistic
+regression, the probability that a data point \( x_i \)
+belongs to a category \( y_i=\{0,1\} \) is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event,
+$$
+p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}.
+$$
-
+Note that \( 1-p(t)= p(-t) \).
-
-
"""The sigmoid function (or the logistic curve) is a
-function that takes any real number, z, and outputs a number (0,1).
-It is useful in neural networks for assigning weights on a relative scale.
-The value z is the weighted sum of parameters involved in the learning algorithm."""
-
-import numpy
-import matplotlib.pyplot as plt
-import math as mt
-
-z = numpy.arange(-5, 5, .1)
-sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))
-sigma = sigma_fn(z)
-
-fig = plt.figure()
-ax = fig.add_subplot(111)
-ax.plot(z, sigma)
-ax.set_ylim([-0.1, 1.1])
-ax.set_xlim([-5,5])
-ax.grid(True)
-ax.set_xlabel('z')
-ax.set_title('sigmoid function')
-
-plt.show()
-
-"""Step Function"""
-z = numpy.arange(-5, 5, .02)
-step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)
-step = step_fn(z)
-
-fig = plt.figure()
-ax = fig.add_subplot(111)
-ax.plot(z, step)
-ax.set_ylim([-0.5, 1.5])
-ax.set_xlim([-5,5])
-ax.grid(True)
-ax.set_xlabel('z')
-ax.set_title('step function')
-
-plt.show()
-
-"""tanh Function"""
-z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)
-t = numpy.tanh(z)
-
-fig = plt.figure()
-ax = fig.add_subplot(111)
-ax.plot(z, t)
-ax.set_ylim([-1.0, 1.0])
-ax.set_xlim([-2*mt.pi,2*mt.pi])
-ax.grid(True)
-ax.set_xlabel('z')
-ax.set_title('tanh function')
-
-plt.show()
-
@@ -321,7 +280,7 @@ plt.show()
34
35
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs026.html b/doc/pub/week38/html/._week38-bs026.html
index 410f91579..529eb67a1 100644
--- a/doc/pub/week38/html/._week38-bs026.html
+++ b/doc/pub/week38/html/._week38-bs026.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,25 +234,69 @@ MathJax.Hub.Config({
-Two parameters
+Examples of likelihood functions used in logistic regression and nueral networks
-We assume now that we have two classes with \( y_i \) either \( 0 \) or \( 1 \). Furthermore we assume also that we have only two parameters \( \beta \) in our fitting of the Sigmoid function, that is we define probabilities
-$$
-\begin{align*}
-p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\
-p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}),
-\end{align*}
-$$
-
-where \( \hat{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).
+The following code plots the logistic function, the step function and other functions we will encounter from here and on.
-Note that we used
-$$
-p(y_i=0\vert x_i, \hat{\beta}) = 1-p(y_i=1\vert x_i, \hat{\beta}).
-$$
+
+
"""The sigmoid function (or the logistic curve) is a
+function that takes any real number, z, and outputs a number (0,1).
+It is useful in neural networks for assigning weights on a relative scale.
+The value z is the weighted sum of parameters involved in the learning algorithm."""
+
+import numpy
+import matplotlib.pyplot as plt
+import math as mt
+
+z = numpy.arange(-5, 5, .1)
+sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))
+sigma = sigma_fn(z)
+
+fig = plt.figure()
+ax = fig.add_subplot(111)
+ax.plot(z, sigma)
+ax.set_ylim([-0.1, 1.1])
+ax.set_xlim([-5,5])
+ax.grid(True)
+ax.set_xlabel('z')
+ax.set_title('sigmoid function')
+
+plt.show()
+
+"""Step Function"""
+z = numpy.arange(-5, 5, .02)
+step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)
+step = step_fn(z)
+
+fig = plt.figure()
+ax = fig.add_subplot(111)
+ax.plot(z, step)
+ax.set_ylim([-0.5, 1.5])
+ax.set_xlim([-5,5])
+ax.grid(True)
+ax.set_xlabel('z')
+ax.set_title('step function')
+
+plt.show()
+
+"""tanh Function"""
+z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)
+t = numpy.tanh(z)
+
+fig = plt.figure()
+ax = fig.add_subplot(111)
+ax.plot(z, t)
+ax.set_ylim([-1.0, 1.0])
+ax.set_xlim([-2*mt.pi,2*mt.pi])
+ax.grid(True)
+ax.set_xlabel('z')
+ax.set_title('tanh function')
+
+plt.show()
+
@@ -277,7 +323,7 @@ $$
35
36
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs027.html b/doc/pub/week38/html/._week38-bs027.html
index 4e39999d5..38b9db190 100644
--- a/doc/pub/week38/html/._week38-bs027.html
+++ b/doc/pub/week38/html/._week38-bs027.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -230,26 +232,25 @@ MathJax.Hub.Config({
-
+
-Maximum likelihood
+Two parameters
-In order to define the total likelihood for all possible outcomes from a
-dataset \( \mathcal{D}=\{(y_i,x_i)\} \), with the binary labels
-\( y_i\in\{0,1\} \) and where the data points are drawn independently, we use the so-called Maximum Likelihood Estimation (MLE) principle.
-We aim thus at maximizing
-the probability of seeing the observed data. We can then approximate the
-likelihood in terms of the product of the individual probabilities of a specific outcome \( y_i \), that is
+We assume now that we have two classes with \( y_i \) either \( 0 \) or \( 1 \). Furthermore we assume also that we have only two parameters \( \beta \) in our fitting of the Sigmoid function, that is we define probabilities
$$
\begin{align*}
-P(\mathcal{D}|\hat{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\hat{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\hat{\beta}))\right]^{1-y_i}\nonumber \\
+p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\
+p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}),
\end{align*}
$$
-from which we obtain the log-likelihood and our cost/loss function
+where \( \hat{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).
+
+
+Note that we used
$$
-\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\hat{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\hat{\beta}))\right]\right).
+p(y_i=0\vert x_i, \hat{\beta}) = 1-p(y_i=1\vert x_i, \hat{\beta}).
$$
@@ -278,7 +279,7 @@ $$
36
37
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs028.html b/doc/pub/week38/html/._week38-bs028.html
index 67b3ff218..1338b682e 100644
--- a/doc/pub/week38/html/._week38-bs028.html
+++ b/doc/pub/week38/html/._week38-bs028.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -230,26 +232,28 @@ MathJax.Hub.Config({
-
+
-The cost function rewritten
+Maximum likelihood
-Reordering the logarithms, we can rewrite the cost/loss function as
+In order to define the total likelihood for all possible outcomes from a
+dataset \( \mathcal{D}=\{(y_i,x_i)\} \), with the binary labels
+\( y_i\in\{0,1\} \) and where the data points are drawn independently, we use the so-called Maximum Likelihood Estimation (MLE) principle.
+We aim thus at maximizing
+the probability of seeing the observed data. We can then approximate the
+likelihood in terms of the product of the individual probabilities of a specific outcome \( y_i \), that is
$$
-\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
+\begin{align*}
+P(\mathcal{D}|\hat{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\hat{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\hat{\beta}))\right]^{1-y_i}\nonumber \\
+\end{align*}
$$
-
-The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to \( \beta \).
-Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that
+from which we obtain the log-likelihood and our cost/loss function
$$
-\mathcal{C}(\hat{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
+\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\hat{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\hat{\beta}))\right]\right).
$$
-This equation is known in statistics as the cross entropy. Finally, we note that just as in linear regression,
-in practice we often supplement the cross-entropy with additional regularization terms, usually \( L_1 \) and \( L_2 \) regularization as we did for Ridge and Lasso regression.
-
@@ -275,6 +279,8 @@ in practice we often supplement the cross-entropy with additional regularization
36
37
38
+ ...
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs029.html b/doc/pub/week38/html/._week38-bs029.html
index 36fc5e03d..1d0de7d76 100644
--- a/doc/pub/week38/html/._week38-bs029.html
+++ b/doc/pub/week38/html/._week38-bs029.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,24 +234,23 @@ MathJax.Hub.Config({
-Minimizing the cross entropy
+The cost function rewritten
-The cross entropy is a convex function of the weights \( \hat{\beta} \) and,
-therefore, any local minimizer is a global minimizer.
+Reordering the logarithms, we can rewrite the cost/loss function as
+$$
+\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
+$$
-Minimizing this
-cost function with respect to the two parameters \( \beta_0 \) and \( \beta_1 \) we obtain
-
+The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to \( \beta \).
+Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that
$$
-\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right),
+\mathcal{C}(\hat{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
$$
-and
-$$
-\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right).
-$$
+This equation is known in statistics as the cross entropy. Finally, we note that just as in linear regression,
+in practice we often supplement the cross-entropy with additional regularization terms, usually \( L_1 \) and \( L_2 \) regularization as we did for Ridge and Lasso regression.
@@ -275,6 +276,7 @@ $$
36
37
38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs030.html b/doc/pub/week38/html/._week38-bs030.html
index 5e807f6a2..da5d094c0 100644
--- a/doc/pub/week38/html/._week38-bs030.html
+++ b/doc/pub/week38/html/._week38-bs030.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,24 +234,23 @@ MathJax.Hub.Config({
-A more compact expression
+Minimizing the cross entropy
-Let us now define a vector \( \hat{y} \) with \( n \) elements \( y_i \), an
-\( n\times p \) matrix \( \hat{X} \) which contains the \( x_i \) values and a
-vector \( \hat{p} \) of fitted probabilities \( p(y_i\vert x_i,\hat{\beta}) \). We can rewrite in a more compact form the first
-derivative of cost function as
-
-$$
-\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right).
-$$
+The cross entropy is a convex function of the weights \( \hat{\beta} \) and,
+therefore, any local minimizer is a global minimizer.
-If we in addition define a diagonal matrix \( \hat{W} \) with elements
-\( p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta}) \), we can obtain a compact expression of the second derivative as
+Minimizing this
+cost function with respect to the two parameters \( \beta_0 \) and \( \beta_1 \) we obtain
$$
-\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}.
+\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right),
+$$
+
+and
+$$
+\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right).
$$
@@ -275,6 +276,7 @@ $$
36
37
38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs031.html b/doc/pub/week38/html/._week38-bs031.html
index 4ae856ec3..ff988575f 100644
--- a/doc/pub/week38/html/._week38-bs031.html
+++ b/doc/pub/week38/html/._week38-bs031.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,17 +234,24 @@ MathJax.Hub.Config({
-Extending to more predictors
+A more compact expression
-Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with \( p \) predictors
+Let us now define a vector \( \hat{y} \) with \( n \) elements \( y_i \), an
+\( n\times p \) matrix \( \hat{X} \) which contains the \( x_i \) values and a
+vector \( \hat{p} \) of fitted probabilities \( p(y_i\vert x_i,\hat{\beta}) \). We can rewrite in a more compact form the first
+derivative of cost function as
+
$$
-\log{ \frac{p(\hat{\beta}\hat{x})}{1-p(\hat{\beta}\hat{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p.
+\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right).
$$
-Here we defined \( \hat{x}=[1,x_1,x_2,\dots,x_p] \) and \( \hat{\beta}=[\beta_0, \beta_1, \dots, \beta_p] \) leading to
+
+If we in addition define a diagonal matrix \( \hat{W} \) with elements
+\( p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta}) \), we can obtain a compact expression of the second derivative as
+
$$
-p(\hat{\beta}\hat{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}.
+\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}.
$$
@@ -267,6 +276,7 @@ $$
36
37
38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs032.html b/doc/pub/week38/html/._week38-bs032.html
index 97026d04f..e7b51bb27 100644
--- a/doc/pub/week38/html/._week38-bs032.html
+++ b/doc/pub/week38/html/._week38-bs032.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,31 +234,19 @@ MathJax.Hub.Config({
-Including more classes
+Extending to more predictors
-Till now we have mainly focused on two classes, the so-called binary
-system. Suppose we wish to extend to \( K \) classes. Let us for the sake
-of simplicity assume we have only two predictors. We have then following model
-
+Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with \( p \) predictors
$$
-\log{\frac{p(C=1\vert x)}{p(K\vert x)}} = \beta_{10}+\beta_{11}x_1,
+\log{ \frac{p(\hat{\beta}\hat{x})}{1-p(\hat{\beta}\hat{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p.
$$
-and
+Here we defined \( \hat{x}=[1,x_1,x_2,\dots,x_p] \) and \( \hat{\beta}=[\beta_0, \beta_1, \dots, \beta_p] \) leading to
$$
-\log{\frac{p(C=2\vert x)}{p(K\vert x)}} = \beta_{20}+\beta_{21}x_1,
+p(\hat{\beta}\hat{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}.
$$
-and so on till the class \( C=K-1 \) class
-$$
-\log{\frac{p(C=K-1\vert x)}{p(K\vert x)}} = \beta_{(K-1)0}+\beta_{(K-1)1}x_1,
-$$
-
-
-and the model is specified in term of \( K-1 \) so-called log-odds or
-logit transformations.
-
@@ -278,6 +268,7 @@ and the model is specified in term of \( K-1 \) so-called log-odds or
36
37
38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs033.html b/doc/pub/week38/html/._week38-bs033.html
index 4ef9dc533..a941b452f 100644
--- a/doc/pub/week38/html/._week38-bs033.html
+++ b/doc/pub/week38/html/._week38-bs033.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,45 +234,30 @@ MathJax.Hub.Config({
-More classes
+Including more classes
-In our discussion of neural networks we will encounter the above again
-in terms of a slightly modified function, the so-called Softmax function.
-
-
-The softmax function is used in various multiclass classification
-methods, such as multinomial logistic regression (also known as
-softmax regression), multiclass linear discriminant analysis, naive
-Bayes classifiers, and artificial neural networks. Specifically, in
-multinomial logistic regression and linear discriminant analysis, the
-input to the function is the result of \( K \) distinct linear functions,
-and the predicted probability for the \( k \)-th class given a sample
-vector \( \hat{x} \) and a weighting vector \( \hat{\beta} \) is (with two
-predictors):
+Till now we have mainly focused on two classes, the so-called binary
+system. Suppose we wish to extend to \( K \) classes. Let us for the sake
+of simplicity assume we have only two predictors. We have then following model
$$
-p(C=k\vert \mathbf {x} )=\frac{\exp{(\beta_{k0}+\beta_{k1}x_1)}}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}}.
+\log{\frac{p(C=1\vert x)}{p(K\vert x)}} = \beta_{10}+\beta_{11}x_1,
$$
-It is easy to extend to more predictors. The final class is
+and
$$
-p(C=K\vert \mathbf {x} )=\frac{1}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}},
+\log{\frac{p(C=2\vert x)}{p(K\vert x)}} = \beta_{20}+\beta_{21}x_1,
+$$
+
+and so on till the class \( C=K-1 \) class
+$$
+\log{\frac{p(C=K-1\vert x)}{p(K\vert x)}} = \beta_{(K-1)0}+\beta_{(K-1)1}x_1,
$$
-and they sum to one. Our earlier discussions were all specialized to
-the case with two classes only. It is easy to see from the above that
-what we derived earlier is compatible with these equations.
-
-
-To find the optimal parameters we would typically use a gradient
-descent method. Newton's method and gradient descent methods are
-discussed in the material on optimization
-methods.
-
-
-This will be discussed next week. Before we develop our own codes for logistic regression, we end this lecture by studying the functionality that Scikit-learn offers.
+and the model is specified in term of \( K-1 \) so-called log-odds or
+logit transformations.
@@ -292,6 +279,7 @@ This will be discussed next week. Before we develop our own codes for logistic r
36
37
38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs034.html b/doc/pub/week38/html/._week38-bs034.html
index 172b68701..ddb01022d 100644
--- a/doc/pub/week38/html/._week38-bs034.html
+++ b/doc/pub/week38/html/._week38-bs034.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,42 +234,46 @@ MathJax.Hub.Config({
-Wisconsin Cancer Data
+More classes
-We show here how we can use a simple regression case on the breast
-cancer data using Logistic regression as our algorithm for
-classification.
+In our discussion of neural networks we will encounter the above again
+in terms of a slightly modified function, the so-called Softmax function.
+The softmax function is used in various multiclass classification
+methods, such as multinomial logistic regression (also known as
+softmax regression), multiclass linear discriminant analysis, naive
+Bayes classifiers, and artificial neural networks. Specifically, in
+multinomial logistic regression and linear discriminant analysis, the
+input to the function is the result of \( K \) distinct linear functions,
+and the predicted probability for the \( k \)-th class given a sample
+vector \( \hat{x} \) and a weighting vector \( \hat{\beta} \) is (with two
+predictors):
-
-
import matplotlib.pyplot as plt
-import numpy as np
-from sklearn.model_selection import train_test_split
-from sklearn.datasets import load_breast_cancer
-from sklearn.linear_model import LogisticRegression
+$$
+p(C=k\vert \mathbf {x} )=\frac{\exp{(\beta_{k0}+\beta_{k1}x_1)}}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}}.
+$$
-# Load the data
-cancer = load_breast_cancer()
+It is easy to extend to more predictors. The final class is
+$$
+p(C=K\vert \mathbf {x} )=\frac{1}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}},
+$$
+
+
+and they sum to one. Our earlier discussions were all specialized to
+the case with two classes only. It is easy to see from the above that
+what we derived earlier is compatible with these equations.
+
+
+To find the optimal parameters we would typically use a gradient
+descent method. Newton's method and gradient descent methods are
+discussed in the material on optimization
+methods.
+
+
+This will be discussed next week. Before we develop our own codes for logistic regression, we end this lecture by studying the functionality that Scikit-learn offers.
-X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
-print(X_train.shape)
-print(X_test.shape)
-# Logistic Regression
-logreg = LogisticRegression(solver='lbfgs')
-logreg.fit(X_train, y_train)
-print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
-#now scale the data
-from sklearn.preprocessing import StandardScaler
-scaler = StandardScaler()
-scaler.fit(X_train)
-X_train_scaled = scaler.transform(X_train)
-X_test_scaled = scaler.transform(X_test)
-# Logistic Regression
-logreg.fit(X_train_scaled, y_train)
-print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
-
@@ -287,6 +293,7 @@ logreg.fit(X_train_scaled, y_train)
36
37
38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs035.html b/doc/pub/week38/html/._week38-bs035.html
index 36565a717..b0212be0b 100644
--- a/doc/pub/week38/html/._week38-bs035.html
+++ b/doc/pub/week38/html/._week38-bs035.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,11 +234,13 @@ MathJax.Hub.Config({
-Using the correlation matrix
+Wisconsin Cancer Data
-In addition to the above scores, we could also study the covariance (and the correlation matrix).
-We use Pandas to compute the correlation matrix.
+We show here how we can use a simple regression case on the breast
+cancer data using Logistic regression as our algorithm for
+classification.
+
@@ -245,35 +249,26 @@ We use Pandas to compute the correlation matrix.
from sklearn.model_selection import train_test_split
from sklearn.datasets import load_breast_cancer
from sklearn.linear_model import LogisticRegression
+
+# Load the data
cancer = load_breast_cancer()
-import pandas as pd
-# Making a data frame
-cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
-fig, axes = plt.subplots(15,2,figsize=(10,20))
-malignant = cancer.data[cancer.target == 0]
-benign = cancer.data[cancer.target == 1]
-ax = axes.ravel()
-
-for i in range(30):
- _, bins = np.histogram(cancer.data[:,i], bins =50)
- ax[i].hist(malignant[:,i], bins = bins, alpha = 0.5)
- ax[i].hist(benign[:,i], bins = bins, alpha = 0.5)
- ax[i].set_title(cancer.feature_names[i])
- ax[i].set_yticks(())
-ax[0].set_xlabel("Feature magnitude")
-ax[0].set_ylabel("Frequency")
-ax[0].legend(["Malignant", "Benign"], loc ="best")
-fig.tight_layout()
-plt.show()
-
-import seaborn as sns
-correlation_matrix = cancerpd.corr().round(1)
-# use the heatmap function from seaborn to plot the correlation matrix
-# annot = True to print the values inside the square
-plt.figure(figsize=(15,8))
-sns.heatmap(data=correlation_matrix, annot=True)
-plt.show()
+X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
+print(X_train.shape)
+print(X_test.shape)
+# Logistic Regression
+logreg = LogisticRegression(solver='lbfgs')
+logreg.fit(X_train, y_train)
+print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
+#now scale the data
+from sklearn.preprocessing import StandardScaler
+scaler = StandardScaler()
+scaler.fit(X_train)
+X_train_scaled = scaler.transform(X_train)
+X_test_scaled = scaler.transform(X_test)
+# Logistic Regression
+logreg.fit(X_train_scaled, y_train)
+print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
@@ -293,6 +288,7 @@ plt.show()
36
37
38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs036.html b/doc/pub/week38/html/._week38-bs036.html
index aa03d6ac8..f2fc821ea 100644
--- a/doc/pub/week38/html/._week38-bs036.html
+++ b/doc/pub/week38/html/._week38-bs036.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,42 +234,49 @@ MathJax.Hub.Config({
-Discussing the correlation data
+Using the correlation matrix
-In the above example we note two things. In the first plot we display
-the overlap of benign and malignant tumors as functions of the various
-features in the Wisconsing breast cancer data set. We see that for
-some of the features we can distinguish clearly the benign and
-malignant cases while for other features we cannot. This can point to
-us which features may be of greater interest when we wish to classify
-a benign or not benign tumour.
-
-
-In the second figure we have computed the so-called correlation
-matrix, which in our case with thirty features becomes a \( 30\times 30 \)
-matrix.
-
-
-We constructed this matrix using pandas via the statements
+In addition to the above scores, we could also study the covariance (and the correlation matrix).
+We use Pandas to compute the correlation matrix.
-
cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
-
-
-and then
-
+
import matplotlib.pyplot as plt
+import numpy as np
+from sklearn.model_selection import train_test_split
+from sklearn.datasets import load_breast_cancer
+from sklearn.linear_model import LogisticRegression
+cancer = load_breast_cancer()
+import pandas as pd
+# Making a data frame
+cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
-
-correlation_matrix = cancerpd.corr().round(1)
-
-
-Diagonalizing this matrix we can in turn say something about which
-features are of relevance and which are not. This leads us to
-the classical Principal Component Analysis (PCA) theorem with
-applications. This will be discussed later this semester (week 43).
+fig, axes = plt.subplots(15,2,figsize=(10,20))
+malignant = cancer.data[cancer.target == 0]
+benign = cancer.data[cancer.target == 1]
+ax = axes.ravel()
+for i in range(30):
+ _, bins = np.histogram(cancer.data[:,i], bins =50)
+ ax[i].hist(malignant[:,i], bins = bins, alpha = 0.5)
+ ax[i].hist(benign[:,i], bins = bins, alpha = 0.5)
+ ax[i].set_title(cancer.feature_names[i])
+ ax[i].set_yticks(())
+ax[0].set_xlabel("Feature magnitude")
+ax[0].set_ylabel("Frequency")
+ax[0].legend(["Malignant", "Benign"], loc ="best")
+fig.tight_layout()
+plt.show()
+
+import seaborn as sns
+correlation_matrix = cancerpd.corr().round(1)
+# use the heatmap function from seaborn to plot the correlation matrix
+# annot = True to print the values inside the square
+plt.figure(figsize=(15,8))
+sns.heatmap(data=correlation_matrix, annot=True)
+plt.show()
+
@@ -285,6 +294,7 @@ applications. This will be discussed later this semester (36
37
38
+ 39
»
diff --git a/doc/pub/week38/html/._week38-bs037.html b/doc/pub/week38/html/._week38-bs037.html
index 41da89372..36cda34bf 100644
--- a/doc/pub/week38/html/._week38-bs037.html
+++ b/doc/pub/week38/html/._week38-bs037.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -232,57 +234,43 @@ MathJax.Hub.Config({
-Other measures in classification studies: Cancer Data again
+Discussing the correlation data
+
+
+In the above example we note two things. In the first plot we display
+the overlap of benign and malignant tumors as functions of the various
+features in the Wisconsing breast cancer data set. We see that for
+some of the features we can distinguish clearly the benign and
+malignant cases while for other features we cannot. This can point to
+us which features may be of greater interest when we wish to classify
+a benign or not benign tumour.
+
+
+In the second figure we have computed the so-called correlation
+matrix, which in our case with thirty features becomes a \( 30\times 30 \)
+matrix.
+
+
+We constructed this matrix using pandas via the statements
-
import matplotlib.pyplot as plt
-import numpy as np
-from sklearn.model_selection import train_test_split
-from sklearn.datasets import load_breast_cancer
-from sklearn.linear_model import LogisticRegression
-
-# Load the data
-cancer = load_breast_cancer()
-
-X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
-print(X_train.shape)
-print(X_test.shape)
-# Logistic Regression
-logreg = LogisticRegression(solver='lbfgs')
-logreg.fit(X_train, y_train)
-print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
-#now scale the data
-from sklearn.preprocessing import StandardScaler
-scaler = StandardScaler()
-scaler.fit(X_train)
-X_train_scaled = scaler.transform(X_train)
-X_test_scaled = scaler.transform(X_test)
-# Logistic Regression
-logreg.fit(X_train_scaled, y_train)
-print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
-
-
-from sklearn.preprocessing import LabelEncoder
-from sklearn.model_selection import cross_validate
-#Cross validation
-accuracy = cross_validate(logreg,X_test_scaled,y_test,cv=10)['test_score']
-print(accuracy)
-print("Test set accuracy with Logistic Regression and scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
-
-
-import scikitplot as skplt
-y_pred = logreg.predict(X_test_scaled)
-skplt.metrics.plot_confusion_matrix(y_test, y_pred, normalize=True)
-plt.show()
-y_probas = logreg.predict_proba(X_test_scaled)
-skplt.metrics.plot_roc(y_test, y_probas)
-plt.show()
-skplt.metrics.plot_cumulative_gain(y_test, y_probas)
-plt.show()
+cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
+and then
+
+
+
correlation_matrix = cancerpd.corr().round(1)
+
+
+Diagonalizing this matrix we can in turn say something about which
+features are of relevance and which are not. This leads us to
+the classical Principal Component Analysis (PCA) theorem with
+applications. This will be discussed later this semester (week 43).
+
+
diff --git a/doc/pub/week38/html/._week38-bs038.html b/doc/pub/week38/html/._week38-bs038.html
index b536c01a4..e143fa415 100644
--- a/doc/pub/week38/html/._week38-bs038.html
+++ b/doc/pub/week38/html/._week38-bs038.html
@@ -42,7 +42,6 @@ Automatically generated HTML file from DocOnce source
Plans for week 38
- Thursday September 17
- Ridge and LASSO Regression, reminder
- Various steps in cross-validation
- How to set up the cross-validation for Ridge and/or Lasso
- Cross-validation in brief
- Code Example for Cross-validation and \( k \)-fold Cross-validation
- Bias-Variance tradeoff with Bootstrap
- Another Example from Scikit-Learn's Repository
- Cross-validation with Ridge
- The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Friday September 18: Intro to Logistic Regression
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ Ridge and LASSO Regression, reminder
+ Various steps in cross-validation
+ How to set up the cross-validation for Ridge and/or Lasso
+ Cross-validation in brief
+ Code Example for Cross-validation and \( k \)-fold Cross-validation
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -251,47 +234,57 @@ MathJax.Hub.Config({
-More classes
-
+Other measures in classification studies: Cancer Data again
-In our discussion of neural networks we will encounter the above again
-in terms of a slightly modified function, the so-called Softmax function.
+
+
import matplotlib.pyplot as plt
+import numpy as np
+from sklearn.model_selection import train_test_split
+from sklearn.datasets import load_breast_cancer
+from sklearn.linear_model import LogisticRegression
+
+# Load the data
+cancer = load_breast_cancer()
+
+X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
+print(X_train.shape)
+print(X_test.shape)
+# Logistic Regression
+logreg = LogisticRegression(solver='lbfgs')
+logreg.fit(X_train, y_train)
+print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
+#now scale the data
+from sklearn.preprocessing import StandardScaler
+scaler = StandardScaler()
+scaler.fit(X_train)
+X_train_scaled = scaler.transform(X_train)
+X_test_scaled = scaler.transform(X_test)
+# Logistic Regression
+logreg.fit(X_train_scaled, y_train)
+print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
+
+
+from sklearn.preprocessing import LabelEncoder
+from sklearn.model_selection import cross_validate
+#Cross validation
+accuracy = cross_validate(logreg,X_test_scaled,y_test,cv=10)['test_score']
+print(accuracy)
+print("Test set accuracy with Logistic Regression and scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
+
+
+import scikitplot as skplt
+y_pred = logreg.predict(X_test_scaled)
+skplt.metrics.plot_confusion_matrix(y_test, y_pred, normalize=True)
+plt.show()
+y_probas = logreg.predict_proba(X_test_scaled)
+skplt.metrics.plot_roc(y_test, y_probas)
+plt.show()
+skplt.metrics.plot_cumulative_gain(y_test, y_probas)
+plt.show()
+
-The softmax function is used in various multiclass classification
-methods, such as multinomial logistic regression (also known as
-softmax regression), multiclass linear discriminant analysis, naive
-Bayes classifiers, and artificial neural networks. Specifically, in
-multinomial logistic regression and linear discriminant analysis, the
-input to the function is the result of \( K \) distinct linear functions,
-and the predicted probability for the \( k \)-th class given a sample
-vector \( \hat{x} \) and a weighting vector \( \hat{\beta} \) is (with two
-predictors):
-$$
-p(C=k\vert \mathbf {x} )=\frac{\exp{(\beta_{k0}+\beta_{k1}x_1)}}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}}.
-$$
-
-It is easy to extend to more predictors. The final class is
-$$
-p(C=K\vert \mathbf {x} )=\frac{1}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}},
-$$
-
-
-and they sum to one. Our earlier discussions were all specialized to
-the case with two classes only. It is easy to see from the above that
-what we derived earlier is compatible with these equations.
-
-
-To find the optimal parameters we would typically use a gradient
-descent method. Newton's method and gradient descent methods are
-discussed in the material on optimization
-methods.
-
-
-This will be discussed next week. Before we develop our own codes for logistic regression, we end this lecture by studying the functionality that Scikit-learn offers.
-
-
@@ -307,11 +300,6 @@ This will be discussed next week. Before we develop our own codes for logistic r
- 37
- 38
- 39
- - 40
- - 41
- - 42
- - 43
- - »
diff --git a/doc/pub/week38/html/week38-bs.html b/doc/pub/week38/html/week38-bs.html
index 79aaab611..f40e5396c 100644
--- a/doc/pub/week38/html/week38-bs.html
+++ b/doc/pub/week38/html/week38-bs.html
@@ -63,6 +63,7 @@ Automatically generated HTML file from DocOnce source
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -186,37 +187,38 @@ MathJax.Hub.Config({
How to set up the cross-validation for Ridge and/or Lasso
Cross-validation in brief
Code Example for Cross-validation and \( k \)-fold Cross-validation
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
+ To think about
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
@@ -275,7 +277,7 @@ MathJax.Hub.Config({
9
10
...
- 38
+ 39
»
diff --git a/doc/pub/week38/html/week38-reveal.html b/doc/pub/week38/html/week38-reveal.html
index 05865da0f..e822d9758 100644
--- a/doc/pub/week38/html/week38-reveal.html
+++ b/doc/pub/week38/html/week38-reveal.html
@@ -418,6 +418,105 @@ plt.show()
+
+To think about
+
+
+When you are comparing your own code with for example Scikit-Learn's library, there are some minor things to keep in mind.
+The example here shows how one can leave out or keep the intercept.
+
+
+
+
+
import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.model_selection import train_test_split
+from sklearn import linear_model
+
+def R2(y_data, y_model):
+ return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
+
+
+# A seed just to ensure that the random numbers are the same for every run.
+# Useful for eventual debugging.
+np.random.seed(3155)
+
+n = 100
+x = np.random.rand(n)
+y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
+
+Maxpolydegree = 20
+X = np.zeros((n,Maxpolydegree))
+X[:,0] = 1.0
+
+for polydegree in range(1, Maxpolydegree):
+ for degree in range(polydegree):
+ X[:,degree] = x**degree
+
+
+# We split the data in test and training data
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
+
+# matrix inversion to find beta
+OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train
+print(OLSbeta)
+# and then make the prediction
+ytildeOLS = X_train @ OLSbeta
+print("Training MSE for OLS")
+print(MSE(y_train,ytildeOLS))
+ypredictOLS = X_test @ OLSbeta
+print("Test MSE OLS")
+print(MSE(y_test,ypredictOLS))
+
+p = len(OLSbeta)
+I = np.eye(p,p)
+# Decide which values of lambda to use
+nlambdas = 4
+MSEOwnRidgePredict = np.zeros(nlambdas)
+MSEOwnRidgeTrain = np.zeros(nlambdas)
+MSERidgePredict = np.zeros(nlambdas)
+MSERidgeTrain = np.zeros(nlambdas)
+
+lambdas = np.logspace(-4, 4, nlambdas)
+for i in range(nlambdas):
+ lmb = lambdas[i]
+ OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
+ # include lasso using Scikit-Learn
+ # Note: we include the intercept
+ RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
+ RegRidge.fit(X_train,y_train)
+ # and then make the prediction
+ ytildeOwnRidge = X_train @ OwnRidgeBeta
+ ypredictOwnRidge = X_test @ OwnRidgeBeta
+ ytildeRidge = RegRidge.predict(X_train)
+ ypredictRidge = RegRidge.predict(X_test)
+ MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
+ MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)
+ MSERidgePredict[i] = MSE(y_test,ypredictRidge)
+ MSERidgeTrain[i] = MSE(y_train,ytildeRidge)
+ print("Beta values for own Ridge implementation")
+ print(OwnRidgeBeta)
+ print("Beta values for Scikit-Learn Ridge implementation")
+ print(RegRidge.coef_)
+# Now plot the results
+plt.figure()
+plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r--', label = 'MSE Ridge Test')
+plt.plot(np.log10(lambdas), MSERidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSERidgePredict, 'r--', label = 'MSE Ridge Test')
+
+plt.xlabel('log10(lambda)')
+plt.ylabel('MSE')
+plt.legend()
+plt.show()
+
+
+
+
More complicated Example: The Ising model
diff --git a/doc/pub/week38/html/week38-solarized.html b/doc/pub/week38/html/week38-solarized.html
index f8998f806..10c67f7b6 100644
--- a/doc/pub/week38/html/week38-solarized.html
+++ b/doc/pub/week38/html/week38-solarized.html
@@ -57,6 +57,7 @@ div { text-align: justify; text-justify: inter-word; }
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -425,6 +426,104 @@ plt.show()
+
To think about
+
+
+When you are comparing your own code with for example Scikit-Learn's library, there are some minor things to keep in mind.
+The example here shows how one can leave out or keep the intercept.
+
+
+
+
+
import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.model_selection import train_test_split
+from sklearn import linear_model
+
+def R2(y_data, y_model):
+ return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
+
+
+# A seed just to ensure that the random numbers are the same for every run.
+# Useful for eventual debugging.
+np.random.seed(3155)
+
+n = 100
+x = np.random.rand(n)
+y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
+
+Maxpolydegree = 20
+X = np.zeros((n,Maxpolydegree))
+X[:,0] = 1.0
+
+for polydegree in range(1, Maxpolydegree):
+ for degree in range(polydegree):
+ X[:,degree] = x**degree
+
+
+# We split the data in test and training data
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
+
+# matrix inversion to find beta
+OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train
+print(OLSbeta)
+# and then make the prediction
+ytildeOLS = X_train @ OLSbeta
+print("Training MSE for OLS")
+print(MSE(y_train,ytildeOLS))
+ypredictOLS = X_test @ OLSbeta
+print("Test MSE OLS")
+print(MSE(y_test,ypredictOLS))
+
+p = len(OLSbeta)
+I = np.eye(p,p)
+# Decide which values of lambda to use
+nlambdas = 4
+MSEOwnRidgePredict = np.zeros(nlambdas)
+MSEOwnRidgeTrain = np.zeros(nlambdas)
+MSERidgePredict = np.zeros(nlambdas)
+MSERidgeTrain = np.zeros(nlambdas)
+
+lambdas = np.logspace(-4, 4, nlambdas)
+for i in range(nlambdas):
+ lmb = lambdas[i]
+ OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
+ # include lasso using Scikit-Learn
+ # Note: we include the intercept
+ RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
+ RegRidge.fit(X_train,y_train)
+ # and then make the prediction
+ ytildeOwnRidge = X_train @ OwnRidgeBeta
+ ypredictOwnRidge = X_test @ OwnRidgeBeta
+ ytildeRidge = RegRidge.predict(X_train)
+ ypredictRidge = RegRidge.predict(X_test)
+ MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
+ MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)
+ MSERidgePredict[i] = MSE(y_test,ypredictRidge)
+ MSERidgeTrain[i] = MSE(y_train,ytildeRidge)
+ print("Beta values for own Ridge implementation")
+ print(OwnRidgeBeta)
+ print("Beta values for Scikit-Learn Ridge implementation")
+ print(RegRidge.coef_)
+# Now plot the results
+plt.figure()
+plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r--', label = 'MSE Ridge Test')
+plt.plot(np.log10(lambdas), MSERidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSERidgePredict, 'r--', label = 'MSE Ridge Test')
+
+plt.xlabel('log10(lambda)')
+plt.ylabel('MSE')
+plt.legend()
+plt.show()
+
+
+
+
More complicated Example: The Ising model
diff --git a/doc/pub/week38/html/week38.html b/doc/pub/week38/html/week38.html
index 8a7d84238..90b0585ae 100644
--- a/doc/pub/week38/html/week38.html
+++ b/doc/pub/week38/html/week38.html
@@ -62,6 +62,7 @@ div { text-align: justify; text-justify: inter-word; }
2,
None,
'code-example-for-cross-validation-and-k-fold-cross-validation'),
+ ('To think about', 2, None, 'to-think-about'),
('More complicated Example: The Ising model',
2,
None,
@@ -430,6 +431,104 @@ plt.show()
+
To think about
+
+
+When you are comparing your own code with for example Scikit-Learn's library, there are some minor things to keep in mind.
+The example here shows how one can leave out or keep the intercept.
+
+
+
+
+
import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.model_selection import train_test_split
+from sklearn import linear_model
+
+def R2(y_data, y_model):
+ return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
+
+
+# A seed just to ensure that the random numbers are the same for every run.
+# Useful for eventual debugging.
+np.random.seed(3155)
+
+n = 100
+x = np.random.rand(n)
+y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
+
+Maxpolydegree = 20
+X = np.zeros((n,Maxpolydegree))
+X[:,0] = 1.0
+
+for polydegree in range(1, Maxpolydegree):
+ for degree in range(polydegree):
+ X[:,degree] = x**degree
+
+
+# We split the data in test and training data
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
+
+# matrix inversion to find beta
+OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train
+print(OLSbeta)
+# and then make the prediction
+ytildeOLS = X_train @ OLSbeta
+print("Training MSE for OLS")
+print(MSE(y_train,ytildeOLS))
+ypredictOLS = X_test @ OLSbeta
+print("Test MSE OLS")
+print(MSE(y_test,ypredictOLS))
+
+p = len(OLSbeta)
+I = np.eye(p,p)
+# Decide which values of lambda to use
+nlambdas = 4
+MSEOwnRidgePredict = np.zeros(nlambdas)
+MSEOwnRidgeTrain = np.zeros(nlambdas)
+MSERidgePredict = np.zeros(nlambdas)
+MSERidgeTrain = np.zeros(nlambdas)
+
+lambdas = np.logspace(-4, 4, nlambdas)
+for i in range(nlambdas):
+ lmb = lambdas[i]
+ OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
+ # include lasso using Scikit-Learn
+ # Note: we include the intercept
+ RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
+ RegRidge.fit(X_train,y_train)
+ # and then make the prediction
+ ytildeOwnRidge = X_train @ OwnRidgeBeta
+ ypredictOwnRidge = X_test @ OwnRidgeBeta
+ ytildeRidge = RegRidge.predict(X_train)
+ ypredictRidge = RegRidge.predict(X_test)
+ MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
+ MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)
+ MSERidgePredict[i] = MSE(y_test,ypredictRidge)
+ MSERidgeTrain[i] = MSE(y_train,ytildeRidge)
+ print("Beta values for own Ridge implementation")
+ print(OwnRidgeBeta)
+ print("Beta values for Scikit-Learn Ridge implementation")
+ print(RegRidge.coef_)
+# Now plot the results
+plt.figure()
+plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r--', label = 'MSE Ridge Test')
+plt.plot(np.log10(lambdas), MSERidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSERidgePredict, 'r--', label = 'MSE Ridge Test')
+
+plt.xlabel('log10(lambda)')
+plt.ylabel('MSE')
+plt.legend()
+plt.show()
+
+
+
+
More complicated Example: The Ising model
diff --git a/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz b/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz
index 6847525b2..9ae162997 100644
Binary files a/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz and b/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz differ
diff --git a/doc/pub/week38/ipynb/week38.ipynb b/doc/pub/week38/ipynb/week38.ipynb
index fa1b8a6c5..d6dd2c4ec 100644
--- a/doc/pub/week38/ipynb/week38.ipynb
+++ b/doc/pub/week38/ipynb/week38.ipynb
@@ -343,6 +343,112 @@
"plt.show()"
]
},
+ {
+ "cell_type": "markdown",
+ "metadata": {},
+ "source": [
+ "## To think about\n",
+ "\n",
+ "When you are comparing your own code with for example **Scikit-Learn**'s library, there are some minor things to keep in mind.\n",
+ "The example here shows how one can leave out or keep the intercept."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "metadata": {
+ "collapsed": false,
+ "editable": true
+ },
+ "outputs": [],
+ "source": [
+ "import numpy as np\n",
+ "import pandas as pd\n",
+ "import matplotlib.pyplot as plt\n",
+ "from sklearn.model_selection import train_test_split\n",
+ "from sklearn import linear_model\n",
+ "\n",
+ "def R2(y_data, y_model):\n",
+ " return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)\n",
+ "def MSE(y_data,y_model):\n",
+ " n = np.size(y_model)\n",
+ " return np.sum((y_data-y_model)**2)/n\n",
+ "\n",
+ "\n",
+ "# A seed just to ensure that the random numbers are the same for every run.\n",
+ "# Useful for eventual debugging.\n",
+ "np.random.seed(3155)\n",
+ "\n",
+ "n = 100\n",
+ "x = np.random.rand(n)\n",
+ "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)\n",
+ "\n",
+ "Maxpolydegree = 20\n",
+ "X = np.zeros((n,Maxpolydegree))\n",
+ "X[:,0] = 1.0\n",
+ "\n",
+ "for polydegree in range(1, Maxpolydegree):\n",
+ " for degree in range(polydegree):\n",
+ " X[:,degree] = x**degree\n",
+ "\n",
+ "\n",
+ "# We split the data in test and training data\n",
+ "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n",
+ "\n",
+ "# matrix inversion to find beta\n",
+ "OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train\n",
+ "print(OLSbeta)\n",
+ "# and then make the prediction\n",
+ "ytildeOLS = X_train @ OLSbeta\n",
+ "print(\"Training MSE for OLS\")\n",
+ "print(MSE(y_train,ytildeOLS))\n",
+ "ypredictOLS = X_test @ OLSbeta\n",
+ "print(\"Test MSE OLS\")\n",
+ "print(MSE(y_test,ypredictOLS))\n",
+ "\n",
+ "p = len(OLSbeta)\n",
+ "I = np.eye(p,p)\n",
+ "# Decide which values of lambda to use\n",
+ "nlambdas = 4\n",
+ "MSEOwnRidgePredict = np.zeros(nlambdas)\n",
+ "MSEOwnRidgeTrain = np.zeros(nlambdas)\n",
+ "MSERidgePredict = np.zeros(nlambdas)\n",
+ "MSERidgeTrain = np.zeros(nlambdas)\n",
+ "\n",
+ "lambdas = np.logspace(-4, 4, nlambdas)\n",
+ "for i in range(nlambdas):\n",
+ " lmb = lambdas[i]\n",
+ " OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train\n",
+ " # include lasso using Scikit-Learn\n",
+ " # Note: we include the intercept\n",
+ " RegRidge = linear_model.Ridge(lmb,fit_intercept=False)\n",
+ " RegRidge.fit(X_train,y_train)\n",
+ " # and then make the prediction\n",
+ " ytildeOwnRidge = X_train @ OwnRidgeBeta\n",
+ " ypredictOwnRidge = X_test @ OwnRidgeBeta\n",
+ " ytildeRidge = RegRidge.predict(X_train)\n",
+ " ypredictRidge = RegRidge.predict(X_test)\n",
+ " MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)\n",
+ " MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)\n",
+ " MSERidgePredict[i] = MSE(y_test,ypredictRidge)\n",
+ " MSERidgeTrain[i] = MSE(y_train,ytildeRidge)\n",
+ " print(\"Beta values for own Ridge implementation\")\n",
+ " print(OwnRidgeBeta)\n",
+ " print(\"Beta values for Scikit-Learn Ridge implementation\")\n",
+ " print(RegRidge.coef_)\n",
+ "# Now plot the results\n",
+ "plt.figure()\n",
+ "plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, label = 'MSE Ridge train')\n",
+ "plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r--', label = 'MSE Ridge Test')\n",
+ "plt.plot(np.log10(lambdas), MSERidgeTrain, label = 'MSE Ridge train')\n",
+ "plt.plot(np.log10(lambdas), MSERidgePredict, 'r--', label = 'MSE Ridge Test')\n",
+ "\n",
+ "plt.xlabel('log10(lambda)')\n",
+ "plt.ylabel('MSE')\n",
+ "plt.legend()\n",
+ "plt.show()"
+ ]
+ },
{
"cell_type": "markdown",
"metadata": {},
diff --git a/doc/src/week36/test.py b/doc/src/week36/test.py
new file mode 100644
index 000000000..7513d63b4
--- /dev/null
+++ b/doc/src/week36/test.py
@@ -0,0 +1,86 @@
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.model_selection import train_test_split
+from sklearn import linear_model
+
+def R2(y_data, y_model):
+ return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
+
+
+# A seed just to ensure that the random numbers are the same for every run.
+# Useful for eventual debugging.
+np.random.seed(3155)
+
+n = 100
+x = np.random.rand(n)
+y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
+
+Maxpolydegree = 20
+X = np.zeros((n,Maxpolydegree))
+X[:,0] = 1.0
+
+for polydegree in range(1, Maxpolydegree):
+ for degree in range(polydegree):
+ X[:,degree] = x**degree
+
+
+# We split the data in test and training data
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
+
+# matrix inversion to find beta
+OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train
+print(OLSbeta)
+# and then make the prediction
+ytildeOLS = X_train @ OLSbeta
+print("Training MSE for OLS")
+print(MSE(y_train,ytildeOLS))
+ypredictOLS = X_test @ OLSbeta
+print("Test MSE OLS")
+print(MSE(y_test,ypredictOLS))
+
+p = len(OLSbeta)
+I = np.eye(p,p)
+# Decide which values of lambda to use
+nlambdas = 4
+MSEOwnRidgePredict = np.zeros(nlambdas)
+MSEOwnRidgeTrain = np.zeros(nlambdas)
+MSERidgePredict = np.zeros(nlambdas)
+MSERidgeTrain = np.zeros(nlambdas)
+
+lambdas = np.logspace(-4, 4, nlambdas)
+for i in range(nlambdas):
+ lmb = lambdas[i]
+ OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
+ # include lasso using Scikit-Learn
+ # Note: we include the intercept
+ RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
+ RegRidge.fit(X_train,y_train)
+ # and then make the prediction
+ ytildeOwnRidge = X_train @ OwnRidgeBeta
+ ypredictOwnRidge = X_test @ OwnRidgeBeta
+ ytildeRidge = RegRidge.predict(X_train)
+ ypredictRidge = RegRidge.predict(X_test)
+ MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
+ MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)
+ MSERidgePredict[i] = MSE(y_test,ypredictRidge)
+ MSERidgeTrain[i] = MSE(y_train,ytildeRidge)
+ print("Beta values for own Ridge implementation")
+ print(OwnRidgeBeta)
+ print("Beta values for Scikit-Learn Ridge implementation")
+ print(RegRidge.coef_)
+# Now plot the results
+plt.figure()
+plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r--', label = 'MSE Ridge Test')
+plt.plot(np.log10(lambdas), MSERidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSERidgePredict, 'r--', label = 'MSE Ridge Test')
+
+plt.xlabel('log10(lambda)')
+plt.ylabel('MSE')
+plt.legend()
+plt.show()
+
diff --git a/doc/src/week38/week38.do.txt b/doc/src/week38/week38.do.txt
index 4a4749661..0b6311775 100644
--- a/doc/src/week38/week38.do.txt
+++ b/doc/src/week38/week38.do.txt
@@ -233,6 +233,100 @@ plt.show()
!ec
+!split
+===== To think about =====
+
+When you are comparing your own code with for example _Scikit-Learn_'s library, there are some minor things to keep in mind.
+The example here shows how one can leave out or keep the intercept.
+
+!bc pycod
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.model_selection import train_test_split
+from sklearn import linear_model
+
+def R2(y_data, y_model):
+ return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
+
+
+# A seed just to ensure that the random numbers are the same for every run.
+# Useful for eventual debugging.
+np.random.seed(3155)
+
+n = 100
+x = np.random.rand(n)
+y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
+
+Maxpolydegree = 20
+X = np.zeros((n,Maxpolydegree))
+X[:,0] = 1.0
+
+for polydegree in range(1, Maxpolydegree):
+ for degree in range(polydegree):
+ X[:,degree] = x**degree
+
+
+# We split the data in test and training data
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
+
+# matrix inversion to find beta
+OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train
+print(OLSbeta)
+# and then make the prediction
+ytildeOLS = X_train @ OLSbeta
+print("Training MSE for OLS")
+print(MSE(y_train,ytildeOLS))
+ypredictOLS = X_test @ OLSbeta
+print("Test MSE OLS")
+print(MSE(y_test,ypredictOLS))
+
+p = len(OLSbeta)
+I = np.eye(p,p)
+# Decide which values of lambda to use
+nlambdas = 4
+MSEOwnRidgePredict = np.zeros(nlambdas)
+MSEOwnRidgeTrain = np.zeros(nlambdas)
+MSERidgePredict = np.zeros(nlambdas)
+MSERidgeTrain = np.zeros(nlambdas)
+
+lambdas = np.logspace(-4, 4, nlambdas)
+for i in range(nlambdas):
+ lmb = lambdas[i]
+ OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
+ # include lasso using Scikit-Learn
+ # Note: we include the intercept
+ RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
+ RegRidge.fit(X_train,y_train)
+ # and then make the prediction
+ ytildeOwnRidge = X_train @ OwnRidgeBeta
+ ypredictOwnRidge = X_test @ OwnRidgeBeta
+ ytildeRidge = RegRidge.predict(X_train)
+ ypredictRidge = RegRidge.predict(X_test)
+ MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
+ MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)
+ MSERidgePredict[i] = MSE(y_test,ypredictRidge)
+ MSERidgeTrain[i] = MSE(y_train,ytildeRidge)
+ print("Beta values for own Ridge implementation")
+ print(OwnRidgeBeta)
+ print("Beta values for Scikit-Learn Ridge implementation")
+ print(RegRidge.coef_)
+# Now plot the results
+plt.figure()
+plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r--', label = 'MSE Ridge Test')
+plt.plot(np.log10(lambdas), MSERidgeTrain, label = 'MSE Ridge train')
+plt.plot(np.log10(lambdas), MSERidgePredict, 'r--', label = 'MSE Ridge Test')
+
+plt.xlabel('log10(lambda)')
+plt.ylabel('MSE')
+plt.legend()
+plt.show()
+
+!ec
!split
===== More complicated Example: The Ising model =====