diff --git a/doc/pub/week38/html/._week38-bs000.html b/doc/pub/week38/html/._week38-bs000.html index 406c60fa3..d5042016a 100644 --- a/doc/pub/week38/html/._week38-bs000.html +++ b/doc/pub/week38/html/._week38-bs000.html @@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source 2, None, 'linear-regression-code-intercept-handling-first'), - ('Another straight Line, the Intercept and Scaling again', - 2, - None, - 'another-straight-line-the-intercept-and-scaling-again'), ('Code Examples', 2, None, 'code-examples'), ('Taking out the mean', 2, None, 'taking-out-the-mean'), ('More complicated Example: The Ising model', @@ -331,83 +327,82 @@ MathJax.Hub.Config({
-What does this mean for thus? And why do we insist on all this? Let us look at some examples. +What does this mean? And why do we insist on all this? Let us look at some examples.
@@ -471,7 +466,7 @@ What does this mean for thus? And why do we insist on all this? Let us look at s
+The intercept is the value of our output/target variable +when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). + +
+Printing the MSE, we see first that both methods give the same +MSE. However, changing the value of the intercept gives a larger or +smaller MSE, meaning that the MSE is penalized by the value of the +intercept! Setting the intercept to true in the Scikit-Learn +function or simply scaling our results, leads to a fit which is +independent of the specific value of the intercept. +
@@ -528,7 +551,7 @@ plt.show()
-As we said above, the intercept is the value of our output/target variable -when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case). -Let us now study a case where we again fit a straight line including the intercept in the design matrix -using ordinary least squares. We are only interested in the MSE value here. +Armed with this wisdom, we attempt first simply set the intercept eqault to False in our implementation of Ridge regression for yet another vanilla data set.
@@ -444,10 +436,6 @@ using ordinary least squares. We are only interested in the MSE value here. n = np.size(y_model) return np.sum((y_data-y_model)**2)/n -def fit_beta(X, y): - return np.linalg.pinv(X.T @ X) @ X.T @ y - - # A seed just to ensure that the random numbers are the same for every run. # Useful for eventual debugging. @@ -455,36 +443,55 @@ np.random.= 100 x = np.random.rand(n) -y = 10.0+5*x +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) - -Maxpolydegree = 2 +Maxpolydegree = 20 X = np.zeros((n,Maxpolydegree)) - +#We include explicitely the intercept column for degree in range(Maxpolydegree): X[:,degree] = x**degree - - # We split the data in test and training data X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2) -# Intercept is included in the design matrix -# own code first -OwnBeta = fit_beta(X_train, y_train) -# Scikit-Learn -skl = LinearRegression(fit_intercept=False).fit(X_train, y_train) -ypredictOwn = X_test @ OwnBeta -ypredictSKL = skl.predict(X_test) -print(MSE(y_test,ypredictOwn) -print(MSE(y_test,ypredictSKL) +p = Maxpolydegree +I = np.eye(p,p) +# Decide which values of lambda to use +nlambdas = 4 +MSEOwnRidgePredict = np.zeros(nlambdas) +MSERidgePredict = np.zeros(nlambdas) +lambdas = np.logspace(-4, 4, nlambdas) +for i in range(nlambdas): + lmb = lambdas[i] + OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train + # include lasso using Scikit-Learn + # Note: we include the intercept column and no scaling + RegRidge = linear_model.Ridge(lmb,fit_intercept=False) + RegRidge.fit(X_train,y_train) + # and then make the prediction + ytildeOwnRidge = X_train @ OwnRidgeBeta + ypredictOwnRidge = X_test @ OwnRidgeBeta + ytildeRidge = RegRidge.predict(X_train) + ypredictRidge = RegRidge.predict(X_test) + MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge) + MSERidgePredict[i] = MSE(y_test,ypredictRidge) + print("Beta values for own Ridge implementation") + print(OwnRidgeBeta) + print("Beta values for Scikit-Learn Ridge implementation") + print(RegRidge.coef_) +# Now plot the results +plt.figure() +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test') +plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test') + +plt.xlabel('log10(lambda)') +plt.ylabel('MSE') +plt.legend() +plt.show()
-Printing the MSE, we see first that both methods give the same -MSE. However, changing the value of the intercept gives a larger or -smaller MSE, meaning that the MSE is penalized by the value of the -intercept! Setting the intercept to true in the Scikit-Learn -function or simply scaling our results, leads to a fit which is -independent of the specific value of the intercept. +The results here agree when we force Scikit-Learn's Ridge function to include the first column in our design matrix. +We see that the results agree very well. What happens if we do not include the intercept in our fit? +Let us see how we can change this code by zero centering.
@@ -512,7 +519,7 @@ independent of the specific value of the intercept.
-Armed with this wisdom, we attempt first simply set the intercept eqault to False in our implementation of Ridge regression for yet another vanilla data set. - +
@@ -436,68 +427,87 @@ Armed with this wisdom, we attempt first simply set the intercept eqault to F import matplotlib.pyplot as plt from sklearn.model_selection import train_test_split from sklearn import linear_model +from sklearn.preprocessing import StandardScaler def MSE(y_data,y_model): n = np.size(y_model) return np.sum((y_data-y_model)**2)/n - - # A seed just to ensure that the random numbers are the same for every run. # Useful for eventual debugging. -np.random.seed(3155) +np.random.seed(315) n = 100 x = np.random.rand(n) y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) Maxpolydegree = 20 -X = np.zeros((n,Maxpolydegree)) -#We include explicitely the intercept column -for degree in range(Maxpolydegree): - X[:,degree] = x**degree +X = np.zeros((n,Maxpolydegree-1)) + +for degree in range(1,Maxpolydegree): #No intercept column + X[:,degree-1] = x**(degree) + # We split the data in test and training data X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2) -p = Maxpolydegree + + + + +#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable +X_train_mean = np.mean(X_train,axis=0) +#Center by removing mean from each feature +X_train_scaled = X_train - X_train_mean +X_test_scaled = X_test - X_train_mean +#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered) +#Remove the intercept from the training data. +y_scaler = np.mean(y_train) +y_train_scaled = y_train - y_scaler + + +p = Maxpolydegree-1 I = np.eye(p,p) # Decide which values of lambda to use nlambdas = 4 MSEOwnRidgePredict = np.zeros(nlambdas) MSERidgePredict = np.zeros(nlambdas) -lambdas = np.logspace(-4, 4, nlambdas) + +lambdas = np.logspace(-4, 1, nlambdas) for i in range(nlambdas): lmb = lambdas[i] - OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train - # include lasso using Scikit-Learn - # Note: we include the intercept column and no scaling - RegRidge = linear_model.Ridge(lmb,fit_intercept=False) + OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled) + intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data + #Add intercept to prediction + ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ + #EQUIVALENT PREDICTION: + #Add intercept to prediction + ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler + print("Values for own Ridge prediction") + print(ypredictOwnRidge) + RegRidge = linear_model.Ridge(lmb) RegRidge.fit(X_train,y_train) - # and then make the prediction - ytildeOwnRidge = X_train @ OwnRidgeBeta - ypredictOwnRidge = X_test @ OwnRidgeBeta - ytildeRidge = RegRidge.predict(X_train) ypredictRidge = RegRidge.predict(X_test) + print("Values for SL Ridge prediction") + print(ypredictRidge) MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge) MSERidgePredict[i] = MSE(y_test,ypredictRidge) print("Beta values for own Ridge implementation") - print(OwnRidgeBeta) + print(OwnRidgeBeta) #Intercept is given by mean of target variable print("Beta values for Scikit-Learn Ridge implementation") print(RegRidge.coef_) + print('Intercept from own implementation:') + print(intercept_) + print('Intercept from Scikit-Learn Ridge implementation') + print(RegRidge.intercept_) + # Now plot the results plt.figure() -plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test') -plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test') - +plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test') +plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test') plt.xlabel('log10(lambda)') plt.ylabel('MSE') plt.legend() plt.show() -
-The results here agree when we force Scikit-Learn's Ridge function to include the first column in our design matrix. -We see that the results agree very well. What happens if we do not include the intercept in our fit? -Let us see how we can change this code by zero centering. -
@@ -524,7 +534,7 @@ Let us see how we can change this code by zero centering.
+The one-dimensional Ising model with nearest neighbor interaction, no +external field and a constant coupling constant \( J \) is given by + +$$ +\begin{align} + H = -J \sum_{k}^L s_k s_{k + 1}, +\tag{1} +\end{align} +$$ + +
+where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins +in the system is determined by \( L \). For the one-dimensional system +there is no phase transition. + +
+We will look at a system of \( L = 40 \) spins with a coupling constant of +\( J = 1 \). To get enough training data we will generate 10000 states +with their respective energies. +
import numpy as np
-import pandas as pd
import matplotlib.pyplot as plt
+from mpl_toolkits.axes_grid1 import make_axes_locatable
+import seaborn as sns
+import scipy.linalg as scl
from sklearn.model_selection import train_test_split
-from sklearn import linear_model
-from sklearn.preprocessing import StandardScaler
+import tqdm
+sns.set(color_codes=True)
+cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
-def MSE(y_data,y_model):
- n = np.size(y_model)
- return np.sum((y_data-y_model)**2)/n
-# A seed just to ensure that the random numbers are the same for every run.
-# Useful for eventual debugging.
-np.random.seed(315)
+L = 40
+n = int(1e4)
-n = 100
-x = np.random.rand(n)
-y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
+spins = np.random.choice([-1, 1], size=(n, L))
+J = 1.0
-Maxpolydegree = 20
-X = np.zeros((n,Maxpolydegree-1))
+energies = np.zeros(n)
-for degree in range(1,Maxpolydegree): #No intercept column
- X[:,degree-1] = x**(degree)
-
-# We split the data in test and training data
-X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
-
-
-
-
-
-#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
-X_train_mean = np.mean(X_train,axis=0)
-#Center by removing mean from each feature
-X_train_scaled = X_train - X_train_mean
-X_test_scaled = X_test - X_train_mean
-#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)
-#Remove the intercept from the training data.
-y_scaler = np.mean(y_train)
-y_train_scaled = y_train - y_scaler
-
-
-p = Maxpolydegree-1
-I = np.eye(p,p)
-# Decide which values of lambda to use
-nlambdas = 4
-MSEOwnRidgePredict = np.zeros(nlambdas)
-MSERidgePredict = np.zeros(nlambdas)
-
-lambdas = np.logspace(-4, 1, nlambdas)
-for i in range(nlambdas):
- lmb = lambdas[i]
- OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
- intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
- #Add intercept to prediction
- ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_
- #EQUIVALENT PREDICTION:
- #Add intercept to prediction
- ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler
- print("Values for own Ridge prediction")
- print(ypredictOwnRidge)
- RegRidge = linear_model.Ridge(lmb)
- RegRidge.fit(X_train,y_train)
- ypredictRidge = RegRidge.predict(X_test)
- print("Values for SL Ridge prediction")
- print(ypredictRidge)
- MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
- MSERidgePredict[i] = MSE(y_test,ypredictRidge)
- print("Beta values for own Ridge implementation")
- print(OwnRidgeBeta) #Intercept is given by mean of target variable
- print("Beta values for Scikit-Learn Ridge implementation")
- print(RegRidge.coef_)
- print('Intercept from own implementation:')
- print(intercept_)
- print('Intercept from Scikit-Learn Ridge implementation')
- print(RegRidge.intercept_)
-
-# Now plot the results
-plt.figure()
-plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
-plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
-plt.xlabel('log10(lambda)')
-plt.ylabel('MSE')
-plt.legend()
-plt.show()
+for i in range(n):
+ energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
+Here we use ordinary least squares +regression to predict the energy for the nearest neighbor +one-dimensional Ising model on a ring, i.e., the endpoints wrap +around. We will use linear regression to fit a value for +the coupling constant to achieve this. +
@@ -539,7 +498,7 @@ plt.show()
-The one-dimensional Ising model with nearest neighbor interaction, no -external field and a constant coupling constant \( J \) is given by +A more general form for the one-dimensional Ising model is $$ \begin{align} - H = -J \sum_{k}^L s_k s_{k + 1}, -\tag{1} + H = - \sum_j^L \sum_k^L s_j s_k J_{jk}. +\tag{2} \end{align} $$
-where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins -in the system is determined by \( L \). For the one-dimensional system -there is no phase transition. +Here we allow for interactions beyond the nearest neighbors and a state dependent +coupling constant. This latter expression can be formulated as +a matrix-product +$$ +\begin{align} + \boldsymbol{H} = \boldsymbol{X} J, +\tag{3} +\end{align} +$$
-We will look at a system of \( L = 40 \) spins with a coupling constant of -\( J = 1 \). To get enough training data we will generate 10000 states -with their respective energies. +where \( X_{jk} = s_j s_k \) and \( J \) is a matrix which consists of the +elements \( -J_{jk} \). This form of writing the energy fits perfectly +with the form utilized in linear regression, that is + +$$ +\begin{align} + \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}, +\tag{4} +\end{align} +$$ + +
+We split the data in training and test data as discussed in the previous example
-
import numpy as np
-import matplotlib.pyplot as plt
-from mpl_toolkits.axes_grid1 import make_axes_locatable
-import seaborn as sns
-import scipy.linalg as scl
-from sklearn.model_selection import train_test_split
-import tqdm
-sns.set(color_codes=True)
-cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
-
-L = 40
-n = int(1e4)
-
-spins = np.random.choice([-1, 1], size=(n, L))
-J = 1.0
-
-energies = np.zeros(n)
-
+X = np.zeros((n, L ** 2))
for i in range(n):
- energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
+ X[i] = np.outer(spins[i], spins[i]).ravel()
+y = energies
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
-
-Here we use ordinary least squares
-regression to predict the energy for the nearest neighbor
-one-dimensional Ising model on a ring, i.e., the endpoints wrap
-around. We will use linear regression to fit a value for
-the coupling constant to achieve this.
-
@@ -503,7 +491,7 @@ the coupling constant to achieve this.
-A more general form for the one-dimensional Ising model is +In the ordinary least squares method we choose the cost function $$ \begin{align} - H = - \sum_j^L \sum_k^L s_j s_k J_{jk}. -\tag{2} + C(\boldsymbol{X}, \boldsymbol{\beta})= \frac{1}{n}\left\{(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})\right\}. +\tag{5} \end{align} $$
-Here we allow for interactions beyond the nearest neighbors and a state dependent -coupling constant. This latter expression can be formulated as -a matrix-product +We then find the extremal point of \( C \) by taking the derivative with respect to \( \boldsymbol{\beta} \) as discussed above. +This yields the expression for \( \boldsymbol{\beta} \) to be + $$ -\begin{align} - \boldsymbol{H} = \boldsymbol{X} J, -\tag{3} -\end{align} + \boldsymbol{\beta} = \frac{\boldsymbol{X}^T \boldsymbol{y}}{\boldsymbol{X}^T \boldsymbol{X}}, $$
-where \( X_{jk} = s_j s_k \) and \( J \) is a matrix which consists of the -elements \( -J_{jk} \). This form of writing the energy fits perfectly -with the form utilized in linear regression, that is - -$$ -\begin{align} - \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}, -\tag{4} -\end{align} -$$ - -
-We split the data in training and test data as discussed in the previous example +which immediately imposes some requirements on \( \boldsymbol{X} \) as there must exist +an inverse of \( \boldsymbol{X}^T \boldsymbol{X} \). If the expression we are modeling contains an +intercept, i.e., a constant term, we must make sure that the +first column of \( \boldsymbol{X} \) consists of \( 1 \). We do this here
-
X = np.zeros((n, L ** 2))
-for i in range(n):
- X[i] = np.outer(spins[i], spins[i]).ravel()
-y = energies
-X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
+X_train_own = np.concatenate(
+ (np.ones(len(X_train))[:, np.newaxis], X_train),
+ axis=1
+)
+X_test_own = np.concatenate(
+ (np.ones(len(X_test))[:, np.newaxis], X_test),
+ axis=1
+)
+
+
+
+
+
def ols_inv(x: np.ndarray, y: np.ndarray) -> np.ndarray:
+ return scl.inv(x.T @ x) @ (x.T @ y)
+beta = ols_inv(X_train_own, y_train)
@@ -496,7 +489,7 @@ X_train, X_test, y_train, y_test = train_tes
-In the ordinary least squares method we choose the cost function +Doing the inversion directly turns out to be a bad idea since the matrix +\( \boldsymbol{X}^T\boldsymbol{X} \) is singular. An alternative approach is to use the singular +value decomposition. Using the definition of the Moore-Penrose +pseudoinverse we can write the equation for \( \boldsymbol{\beta} \) as +$$ + \boldsymbol{\beta} = \boldsymbol{X}^{+}\boldsymbol{y}, +$$ + +
+where the pseudoinverse of \( \boldsymbol{X} \) is given by + +$$ + \boldsymbol{X}^{+} = \frac{\boldsymbol{X}^T}{\boldsymbol{X}^T\boldsymbol{X}}. +$$ + +
+Using singular value decomposition we can decompose the matrix \( \boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma} \boldsymbol{V}^T \), +where \( \boldsymbol{U} \) and \( \boldsymbol{V} \) are orthogonal(unitary) matrices and \( \boldsymbol{\Sigma} \) contains the singular values (more details below). +where \( X^{+} = V\Sigma^{+} U^T \). This reduces the equation for +\( \omega \) to $$ \begin{align} - C(\boldsymbol{X}, \boldsymbol{\beta})= \frac{1}{n}\left\{(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})\right\}. -\tag{5} + \boldsymbol{\beta} = \boldsymbol{V}\boldsymbol{\Sigma}^{+} \boldsymbol{U}^T \boldsymbol{y}. +\tag{6} \end{align} $$
-We then find the extremal point of \( C \) by taking the derivative with respect to \( \boldsymbol{\beta} \) as discussed above. -This yields the expression for \( \boldsymbol{\beta} \) to be - -$$ - \boldsymbol{\beta} = \frac{\boldsymbol{X}^T \boldsymbol{y}}{\boldsymbol{X}^T \boldsymbol{X}}, -$$ - -
-which immediately imposes some requirements on \( \boldsymbol{X} \) as there must exist -an inverse of \( \boldsymbol{X}^T \boldsymbol{X} \). If the expression we are modeling contains an -intercept, i.e., a constant term, we must make sure that the -first column of \( \boldsymbol{X} \) consists of \( 1 \). We do this here +Note that solving this equation by actually doing the pseudoinverse +(which is what we will do) is not a good idea as this operation scales +as \( \mathcal{O}(n^3) \), where \( n \) is the number of elements in a +general matrix. Instead, doing \( QR \)-factorization and solving the +linear system as an equation would reduce this down to +\( \mathcal{O}(n^2) \) operations.
-
X_train_own = np.concatenate(
- (np.ones(len(X_train))[:, np.newaxis], X_train),
- axis=1
-)
-X_test_own = np.concatenate(
- (np.ones(len(X_test))[:, np.newaxis], X_test),
- axis=1
-)
+def ols_svd(x: np.ndarray, y: np.ndarray) -> np.ndarray:
+ u, s, v = scl.svd(x)
+ return v.T @ scl.pinv(scl.diagsvd(s, u.shape[0], v.shape[0])) @ u.T @ y
-
def ols_inv(x: np.ndarray, y: np.ndarray) -> np.ndarray:
- return scl.inv(x.T @ x) @ (x.T @ y)
-beta = ols_inv(X_train_own, y_train)
+beta = ols_svd(X_train_own,y_train)
+
+When extracting the \( J \)-matrix we need to make sure that we remove the intercept, as is done here
+
+
+
+
+
J = beta[1:].reshape(L, L)
+
+
+A way of looking at the coefficients in \( J \) is to plot the matrices as images.
+
+
+
+
+
fig = plt.figure(figsize=(20, 14))
+im = plt.imshow(J, **cmap_args)
+plt.title("OLS", fontsize=18)
+plt.xticks(fontsize=18)
+plt.yticks(fontsize=18)
+cb = fig.colorbar(im)
+cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
+plt.show()
+
+
+It is interesting to note that OLS
+considers both \( J_{j, j + 1} = -0.5 \) and \( J_{j, j - 1} = -0.5 \) as
+valid matrix elements for \( J \).
+In our discussion below on hyperparameters and Ridge and Lasso regression we will see that
+this problem can be removed, partly and only with Lasso regression.
+
+
+In this case our matrix inversion was actually possible. The obvious question now is what is the mathematics behind the SVD?
+
@@ -494,7 +528,7 @@ beta = ols_inv(X_train_own, y_train)
29
30
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs021.html b/doc/pub/week38/html/._week38-bs021.html
index 28a951dc3..10aaddbc0 100644
--- a/doc/pub/week38/html/._week38-bs021.html
+++ b/doc/pub/week38/html/._week38-bs021.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,74 +418,128 @@ MathJax.Hub.Config({
-Singular Value decomposition
+The one-dimensional Ising model
-Doing the inversion directly turns out to be a bad idea since the matrix
-\( \boldsymbol{X}^T\boldsymbol{X} \) is singular. An alternative approach is to use the singular
-value decomposition. Using the definition of the Moore-Penrose
-pseudoinverse we can write the equation for \( \boldsymbol{\beta} \) as
+Let us bring back the Ising model again, but now with an additional
+focus on Ridge and Lasso regression as well. We repeat some of the
+basic parts of the Ising model and the setup of the training and test
+data. The one-dimensional Ising model with nearest neighbor
+interaction, no external field and a constant coupling constant \( J \) is
+given by
-$$
- \boldsymbol{\beta} = \boldsymbol{X}^{+}\boldsymbol{y},
-$$
-
-
-where the pseudoinverse of \( \boldsymbol{X} \) is given by
-
-$$
- \boldsymbol{X}^{+} = \frac{\boldsymbol{X}^T}{\boldsymbol{X}^T\boldsymbol{X}}.
-$$
-
-
-Using singular value decomposition we can decompose the matrix \( \boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma} \boldsymbol{V}^T \),
-where \( \boldsymbol{U} \) and \( \boldsymbol{V} \) are orthogonal(unitary) matrices and \( \boldsymbol{\Sigma} \) contains the singular values (more details below).
-where \( X^{+} = V\Sigma^{+} U^T \). This reduces the equation for
-\( \omega \) to
$$
\begin{align}
- \boldsymbol{\beta} = \boldsymbol{V}\boldsymbol{\Sigma}^{+} \boldsymbol{U}^T \boldsymbol{y}.
-\tag{6}
+ H = -J \sum_{k}^L s_k s_{k + 1},
+\tag{7}
+\end{align}
+$$
+
+where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins in the system is determined by \( L \). For the one-dimensional system there is no phase transition.
+
+
+We will look at a system of \( L = 40 \) spins with a coupling constant of \( J = 1 \). To get enough training data we will generate 10000 states with their respective energies.
+
+
+
+
+
import numpy as np
+import matplotlib.pyplot as plt
+from mpl_toolkits.axes_grid1 import make_axes_locatable
+import seaborn as sns
+import scipy.linalg as scl
+from sklearn.model_selection import train_test_split
+import sklearn.linear_model as skl
+import tqdm
+sns.set(color_codes=True)
+cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
+
+L = 40
+n = int(1e4)
+
+spins = np.random.choice([-1, 1], size=(n, L))
+J = 1.0
+
+energies = np.zeros(n)
+
+for i in range(n):
+ energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
+
+
+A more general form for the one-dimensional Ising model is
+
+$$
+\begin{align}
+ H = - \sum_j^L \sum_k^L s_j s_k J_{jk}.
+\tag{8}
\end{align}
$$
-Note that solving this equation by actually doing the pseudoinverse
-(which is what we will do) is not a good idea as this operation scales
-as \( \mathcal{O}(n^3) \), where \( n \) is the number of elements in a
-general matrix. Instead, doing \( QR \)-factorization and solving the
-linear system as an equation would reduce this down to
-\( \mathcal{O}(n^2) \) operations.
+Here we allow for interactions beyond the nearest neighbors and a more
+adaptive coupling matrix. This latter expression can be formulated as
+a matrix-product on the form
+$$
+\begin{align}
+ H = X J,
+\tag{9}
+\end{align}
+$$
+
+
+where \( X_{jk} = s_j s_k \) and \( J \) is the matrix consisting of the
+elements \( -J_{jk} \). This form of writing the energy fits perfectly
+with the form utilized in linear regression, viz.
+$$
+\begin{align}
+ \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}.
+\tag{10}
+\end{align}
+$$
+
+We organize the data as we did above
+
+
+
+
X = np.zeros((n, L ** 2))
+for i in range(n):
+ X[i] = np.outer(spins[i], spins[i]).ravel()
+y = energies
+X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.96)
+
+X_train_own = np.concatenate(
+ (np.ones(len(X_train))[:, np.newaxis], X_train),
+ axis=1
+)
+
+X_test_own = np.concatenate(
+ (np.ones(len(X_test))[:, np.newaxis], X_test),
+ axis=1
+)
+
+
+We will do all fitting with Scikit-Learn,
-
def ols_svd(x: np.ndarray, y: np.ndarray) -> np.ndarray:
- u, s, v = scl.svd(x)
- return v.T @ scl.pinv(scl.diagsvd(s, u.shape[0], v.shape[0])) @ u.T @ y
+clf = skl.LinearRegression().fit(X_train, y_train)
+When extracting the \( J \)-matrix we make sure to remove the intercept
+
-
beta = ols_svd(X_train_own,y_train)
+J_sk = clf.coef_.reshape(L, L)
-When extracting the \( J \)-matrix we need to make sure that we remove the intercept, as is done here
-
-
-
-
-
J = beta[1:].reshape(L, L)
-
-
-A way of looking at the coefficients in \( J \) is to plot the matrices as images.
-
+And then we plot the results
fig = plt.figure(figsize=(20, 14))
-im = plt.imshow(J, **cmap_args)
-plt.title("OLS", fontsize=18)
+im = plt.imshow(J_sk, **cmap_args)
+plt.title("LinearRegression from Scikit-learn", fontsize=18)
plt.xticks(fontsize=18)
plt.yticks(fontsize=18)
cb = fig.colorbar(im)
@@ -498,14 +547,7 @@ cb.ax.se
plt.show()
-It is interesting to note that OLS
-considers both \( J_{j, j + 1} = -0.5 \) and \( J_{j, j - 1} = -0.5 \) as
-valid matrix elements for \( J \).
-In our discussion below on hyperparameters and Ridge and Lasso regression we will see that
-this problem can be removed, partly and only with Lasso regression.
-
-
-In this case our matrix inversion was actually possible. The obvious question now is what is the mathematics behind the SVD?
+The results perfectly with our previous discussion where we used our own code.
@@ -533,7 +575,7 @@ In this case our matrix inversion was actually possible. The obvious question no
30
31
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs022.html b/doc/pub/week38/html/._week38-bs022.html
index f69710e86..d1b4e1c9a 100644
--- a/doc/pub/week38/html/._week38-bs022.html
+++ b/doc/pub/week38/html/._week38-bs022.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,137 +418,38 @@ MathJax.Hub.Config({
-The one-dimensional Ising model
+Ridge regression
-Let us bring back the Ising model again, but now with an additional
-focus on Ridge and Lasso regression as well. We repeat some of the
-basic parts of the Ising model and the setup of the training and test
-data. The one-dimensional Ising model with nearest neighbor
-interaction, no external field and a constant coupling constant \( J \) is
-given by
+Having explored the ordinary least squares we move on to ridge
+regression. In ridge regression we include a regularizer. This
+involves a new cost function which leads to a new estimate for the
+weights \( \boldsymbol{\beta} \). This results in a penalized regression problem. The
+cost function is given by
$$
\begin{align}
- H = -J \sum_{k}^L s_k s_{k + 1},
-\tag{7}
-\end{align}
-$$
-
-where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins in the system is determined by \( L \). For the one-dimensional system there is no phase transition.
-
-
-We will look at a system of \( L = 40 \) spins with a coupling constant of \( J = 1 \). To get enough training data we will generate 10000 states with their respective energies.
-
-
-
-
-
import numpy as np
-import matplotlib.pyplot as plt
-from mpl_toolkits.axes_grid1 import make_axes_locatable
-import seaborn as sns
-import scipy.linalg as scl
-from sklearn.model_selection import train_test_split
-import sklearn.linear_model as skl
-import tqdm
-sns.set(color_codes=True)
-cmap_args=dict(vmin=-1., vmax=1., cmap='seismic')
-
-L = 40
-n = int(1e4)
-
-spins = np.random.choice([-1, 1], size=(n, L))
-J = 1.0
-
-energies = np.zeros(n)
-
-for i in range(n):
- energies[i] = - J * np.dot(spins[i], np.roll(spins[i], 1))
-
-
-A more general form for the one-dimensional Ising model is
-
-$$
-\begin{align}
- H = - \sum_j^L \sum_k^L s_j s_k J_{jk}.
-\tag{8}
+ C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta}.
+\tag{11}
\end{align}
$$
-
-Here we allow for interactions beyond the nearest neighbors and a more
-adaptive coupling matrix. This latter expression can be formulated as
-a matrix-product on the form
-$$
-\begin{align}
- H = X J,
-\tag{9}
-\end{align}
-$$
-
-
-where \( X_{jk} = s_j s_k \) and \( J \) is the matrix consisting of the
-elements \( -J_{jk} \). This form of writing the energy fits perfectly
-with the form utilized in linear regression, viz.
-$$
-\begin{align}
- \boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}.
-\tag{10}
-\end{align}
-$$
-
-We organize the data as we did above
-
X = np.zeros((n, L ** 2))
-for i in range(n):
- X[i] = np.outer(spins[i], spins[i]).ravel()
-y = energies
-X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.96)
-
-X_train_own = np.concatenate(
- (np.ones(len(X_train))[:, np.newaxis], X_train),
- axis=1
-)
-
-X_test_own = np.concatenate(
- (np.ones(len(X_test))[:, np.newaxis], X_test),
- axis=1
-)
-
-
-We will do all fitting with Scikit-Learn,
-
-
-
-
-
clf = skl.LinearRegression().fit(X_train, y_train)
-
-
-When extracting the \( J \)-matrix we make sure to remove the intercept
-
-
-
-
J_sk = clf.coef_.reshape(L, L)
-
-
-And then we plot the results
-
-
-
-
fig = plt.figure(figsize=(20, 14))
-im = plt.imshow(J_sk, **cmap_args)
-plt.title("LinearRegression from Scikit-learn", fontsize=18)
+_lambda = 0.1
+clf_ridge = skl.Ridge(alpha=_lambda).fit(X_train, y_train)
+J_ridge_sk = clf_ridge.coef_.reshape(L, L)
+fig = plt.figure(figsize=(20, 14))
+im = plt.imshow(J_ridge_sk, **cmap_args)
+plt.title("Ridge from Scikit-learn", fontsize=18)
plt.xticks(fontsize=18)
plt.yticks(fontsize=18)
cb = fig.colorbar(im)
cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
+
plt.show()
-
-The results perfectly with our previous discussion where we used our own code.
-
@@ -580,7 +476,7 @@ The results perfectly with our previous discussion where we used our own code.
31
32
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs023.html b/doc/pub/week38/html/._week38-bs023.html
index 7cf5de9eb..c7b61759f 100644
--- a/doc/pub/week38/html/._week38-bs023.html
+++ b/doc/pub/week38/html/._week38-bs023.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,31 +418,29 @@ MathJax.Hub.Config({
-Ridge regression
+LASSO regression
-Having explored the ordinary least squares we move on to ridge
-regression. In ridge regression we include a regularizer. This
-involves a new cost function which leads to a new estimate for the
-weights \( \boldsymbol{\beta} \). This results in a penalized regression problem. The
-cost function is given by
+In the Least Absolute Shrinkage and Selection Operator (LASSO)-method we get a third cost function.
$$
\begin{align}
- C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta}.
-\tag{11}
+ C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \sqrt{\boldsymbol{\beta}^T\boldsymbol{\beta}}.
+\tag{12}
\end{align}
$$
+
+Finding the extremal point of this cost function is not so straight-forward as in least squares and ridge. We will therefore rely solely on the function ``Lasso`` from Scikit-Learn.
+
-
_lambda = 0.1
-clf_ridge = skl.Ridge(alpha=_lambda).fit(X_train, y_train)
-J_ridge_sk = clf_ridge.coef_.reshape(L, L)
+clf_lasso = skl.Lasso(alpha=_lambda).fit(X_train, y_train)
+J_lasso_sk = clf_lasso.coef_.reshape(L, L)
fig = plt.figure(figsize=(20, 14))
-im = plt.imshow(J_ridge_sk, **cmap_args)
-plt.title("Ridge from Scikit-learn", fontsize=18)
+im = plt.imshow(J_lasso_sk, **cmap_args)
+plt.title("Lasso from Scikit-learn", fontsize=18)
plt.xticks(fontsize=18)
plt.yticks(fontsize=18)
cb = fig.colorbar(im)
@@ -455,6 +448,11 @@ cb.ax.se
plt.show()
+
+It is quite striking how LASSO breaks the symmetry of the coupling
+constant as opposed to ridge and OLS. We get a sparse solution with
+\( J_{j, j + 1} = -1 \).
+
@@ -481,7 +479,7 @@ plt.show()
32
33
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs024.html b/doc/pub/week38/html/._week38-bs024.html
index df865fe1c..7ac38c92b 100644
--- a/doc/pub/week38/html/._week38-bs024.html
+++ b/doc/pub/week38/html/._week38-bs024.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,40 +418,56 @@ MathJax.Hub.Config({
-LASSO regression
+Performance as function of the regularization parameter
-In the Least Absolute Shrinkage and Selection Operator (LASSO)-method we get a third cost function.
-
-$$
-\begin{align}
- C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \sqrt{\boldsymbol{\beta}^T\boldsymbol{\beta}}.
-\tag{12}
-\end{align}
-$$
-
-
-Finding the extremal point of this cost function is not so straight-forward as in least squares and ridge. We will therefore rely solely on the function ``Lasso`` from Scikit-Learn.
+We see how the different models perform for a different set of values for \( \lambda \).
-
clf_lasso = skl.Lasso(alpha=_lambda).fit(X_train, y_train)
-J_lasso_sk = clf_lasso.coef_.reshape(L, L)
-fig = plt.figure(figsize=(20, 14))
-im = plt.imshow(J_lasso_sk, **cmap_args)
-plt.title("Lasso from Scikit-learn", fontsize=18)
-plt.xticks(fontsize=18)
-plt.yticks(fontsize=18)
-cb = fig.colorbar(im)
-cb.ax.set_yticklabels(cb.ax.get_yticklabels(), fontsize=18)
+lambdas = np.logspace(-4, 5, 10)
+
+train_errors = {
+ "ols_sk": np.zeros(lambdas.size),
+ "ridge_sk": np.zeros(lambdas.size),
+ "lasso_sk": np.zeros(lambdas.size)
+}
+
+test_errors = {
+ "ols_sk": np.zeros(lambdas.size),
+ "ridge_sk": np.zeros(lambdas.size),
+ "lasso_sk": np.zeros(lambdas.size)
+}
+
+plot_counter = 1
+
+fig = plt.figure(figsize=(32, 54))
+
+for i, _lambda in enumerate(tqdm.tqdm(lambdas)):
+ for key, method in zip(
+ ["ols_sk", "ridge_sk", "lasso_sk"],
+ [skl.LinearRegression(), skl.Ridge(alpha=_lambda), skl.Lasso(alpha=_lambda)]
+ ):
+ method = method.fit(X_train, y_train)
+
+ train_errors[key][i] = method.score(X_train, y_train)
+ test_errors[key][i] = method.score(X_test, y_test)
+
+ omega = method.coef_.reshape(L, L)
+
+ plt.subplot(10, 5, plot_counter)
+ plt.imshow(omega, **cmap_args)
+ plt.title(r"%s, $\lambda = %.4f$" % (key, _lambda))
+ plot_counter += 1
plt.show()
-It is quite striking how LASSO breaks the symmetry of the coupling
-constant as opposed to ridge and OLS. We get a sparse solution with
-\( J_{j, j + 1} = -1 \).
+We see that LASSO reaches a good solution for low
+values of \( \lambda \), but will "wither" when we increase \( \lambda \) too
+much. Ridge is more stable over a larger range of values for
+\( \lambda \), but eventually also fades away.
@@ -484,7 +495,7 @@ constant as opposed to ridge and OLS. We get a sparse solution with
33
34
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs025.html b/doc/pub/week38/html/._week38-bs025.html
index 02f3d0474..3b5c8f105 100644
--- a/doc/pub/week38/html/._week38-bs025.html
+++ b/doc/pub/week38/html/._week38-bs025.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,56 +418,54 @@ MathJax.Hub.Config({
-Performance as function of the regularization parameter
+Finding the optimal value of \( \lambda \)
-We see how the different models perform for a different set of values for \( \lambda \).
+To determine which value of \( \lambda \) is best we plot the accuracy of
+the models when predicting the training and the testing set. We expect
+the accuracy of the training set to be quite good, but if the accuracy
+of the testing set is much lower this tells us that we might be
+subject to an overfit model. The ideal scenario is an accuracy on the
+testing set that is close to the accuracy of the training set.
-
lambdas = np.logspace(-4, 5, 10)
+fig = plt.figure(figsize=(20, 14))
-train_errors = {
- "ols_sk": np.zeros(lambdas.size),
- "ridge_sk": np.zeros(lambdas.size),
- "lasso_sk": np.zeros(lambdas.size)
+colors = {
+ "ols_sk": "r",
+ "ridge_sk": "y",
+ "lasso_sk": "c"
}
-test_errors = {
- "ols_sk": np.zeros(lambdas.size),
- "ridge_sk": np.zeros(lambdas.size),
- "lasso_sk": np.zeros(lambdas.size)
-}
-
-plot_counter = 1
-
-fig = plt.figure(figsize=(32, 54))
-
-for i, _lambda in enumerate(tqdm.tqdm(lambdas)):
- for key, method in zip(
- ["ols_sk", "ridge_sk", "lasso_sk"],
- [skl.LinearRegression(), skl.Ridge(alpha=_lambda), skl.Lasso(alpha=_lambda)]
- ):
- method = method.fit(X_train, y_train)
-
- train_errors[key][i] = method.score(X_train, y_train)
- test_errors[key][i] = method.score(X_test, y_test)
-
- omega = method.coef_.reshape(L, L)
-
- plt.subplot(10, 5, plot_counter)
- plt.imshow(omega, **cmap_args)
- plt.title(r"%s, $\lambda = %.4f$" % (key, _lambda))
- plot_counter += 1
+for key in train_errors:
+ plt.semilogx(
+ lambdas,
+ train_errors[key],
+ colors[key],
+ label="Train {0}".format(key),
+ linewidth=4.0
+ )
+for key in test_errors:
+ plt.semilogx(
+ lambdas,
+ test_errors[key],
+ colors[key] + "--",
+ label="Test {0}".format(key),
+ linewidth=4.0
+ )
+plt.legend(loc="best", fontsize=18)
+plt.xlabel(r"$\lambda$", fontsize=18)
+plt.ylabel(r"$R^2$", fontsize=18)
+plt.tick_params(labelsize=18)
plt.show()
-We see that LASSO reaches a good solution for low
-values of \( \lambda \), but will "wither" when we increase \( \lambda \) too
-much. Ridge is more stable over a larger range of values for
-\( \lambda \), but eventually also fades away.
+From the above figure we can see that LASSO with \( \lambda = 10^{-2} \)
+achieves a very good accuracy on the test set. This by far surpasses the
+other models for all values of \( \lambda \).
@@ -500,7 +493,7 @@ much. Ridge is more stable over a larger range of values for
34
35
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs026.html b/doc/pub/week38/html/._week38-bs026.html
index 79e3730df..8019c1d72 100644
--- a/doc/pub/week38/html/._week38-bs026.html
+++ b/doc/pub/week38/html/._week38-bs026.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -421,56 +416,22 @@ MathJax.Hub.Config({
-
+
-Finding the optimal value of \( \lambda \)
+Logistic Regression
-To determine which value of \( \lambda \) is best we plot the accuracy of
-the models when predicting the training and the testing set. We expect
-the accuracy of the training set to be quite good, but if the accuracy
-of the testing set is much lower this tells us that we might be
-subject to an overfit model. The ideal scenario is an accuracy on the
-testing set that is close to the accuracy of the training set.
-
-
-
-
-
fig = plt.figure(figsize=(20, 14))
-
-colors = {
- "ols_sk": "r",
- "ridge_sk": "y",
- "lasso_sk": "c"
-}
-
-for key in train_errors:
- plt.semilogx(
- lambdas,
- train_errors[key],
- colors[key],
- label="Train {0}".format(key),
- linewidth=4.0
- )
-
-for key in test_errors:
- plt.semilogx(
- lambdas,
- test_errors[key],
- colors[key] + "--",
- label="Test {0}".format(key),
- linewidth=4.0
- )
-plt.legend(loc="best", fontsize=18)
-plt.xlabel(r"$\lambda$", fontsize=18)
-plt.ylabel(r"$R^2$", fontsize=18)
-plt.tick_params(labelsize=18)
-plt.show()
-
-
-From the above figure we can see that LASSO with \( \lambda = 10^{-2} \)
-achieves a very good accuracy on the test set. This by far surpasses the
-other models for all values of \( \lambda \).
+In linear regression our main interest was centered on learning the
+coefficients of a functional fit (say a polynomial) in order to be
+able to predict the response of a continuous variable on some unseen
+data. The fit to the continuous variable \( y_i \) is based on some
+independent variables \( \boldsymbol{x}_i \). Linear regression resulted in
+analytical expressions for standard ordinary Least Squares or Ridge
+regression (in terms of matrices to invert) for several quantities,
+ranging from the variance and thereby the confidence intervals of the
+parameters \( \boldsymbol{\beta} \) to the mean squared error. If we can invert
+the product of the design matrices, linear regression gives then a
+simple recipe for fitting our data.
@@ -498,7 +459,7 @@ other models for all values of \( \lambda \).
35
36
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs027.html b/doc/pub/week38/html/._week38-bs027.html
index f18d73644..ec191791b 100644
--- a/doc/pub/week38/html/._week38-bs027.html
+++ b/doc/pub/week38/html/._week38-bs027.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,20 +418,25 @@ MathJax.Hub.Config({
-Logistic Regression
+Classification problems
-In linear regression our main interest was centered on learning the
-coefficients of a functional fit (say a polynomial) in order to be
-able to predict the response of a continuous variable on some unseen
-data. The fit to the continuous variable \( y_i \) is based on some
-independent variables \( \boldsymbol{x}_i \). Linear regression resulted in
-analytical expressions for standard ordinary Least Squares or Ridge
-regression (in terms of matrices to invert) for several quantities,
-ranging from the variance and thereby the confidence intervals of the
-parameters \( \boldsymbol{\beta} \) to the mean squared error. If we can invert
-the product of the design matrices, linear regression gives then a
-simple recipe for fitting our data.
+Classification problems, however, are concerned with outcomes taking
+the form of discrete variables (i.e. categories). We may for example,
+on the basis of DNA sequencing for a number of patients, like to find
+out which mutations are important for a certain disease; or based on
+scans of various patients' brains, figure out if there is a tumor or
+not; or given a specific physical system, we'd like to identify its
+state, say whether it is an ordered or disordered system (typical
+situation in solid state physics); or classify the status of a
+patient, whether she/he has a stroke or not and many other similar
+situations.
+
+
+The most common situation we encounter when we apply logistic
+regression is that of two possible outcomes, normally denoted as a
+binary outcome, true or false, positive or negative, success or
+failure etc.
@@ -464,7 +464,7 @@ simple recipe for fitting our data.
36
37
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs028.html b/doc/pub/week38/html/._week38-bs028.html
index 6d5f0796b..c17ab46b3 100644
--- a/doc/pub/week38/html/._week38-bs028.html
+++ b/doc/pub/week38/html/._week38-bs028.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -421,27 +416,25 @@ MathJax.Hub.Config({
-
+
-Classification problems
+Optimization and Deep learning
-Classification problems, however, are concerned with outcomes taking
-the form of discrete variables (i.e. categories). We may for example,
-on the basis of DNA sequencing for a number of patients, like to find
-out which mutations are important for a certain disease; or based on
-scans of various patients' brains, figure out if there is a tumor or
-not; or given a specific physical system, we'd like to identify its
-state, say whether it is an ordered or disordered system (typical
-situation in solid state physics); or classify the status of a
-patient, whether she/he has a stroke or not and many other similar
-situations.
+Logistic regression will also serve as our stepping stone towards
+neural network algorithms and supervised deep learning. For logistic
+learning, the minimization of the cost function leads to a non-linear
+equation in the parameters \( \boldsymbol{\beta} \). The optimization of the
+problem calls therefore for minimization algorithms. This forms the
+bottle neck of all machine learning algorithms, namely how to find
+reliable minima of a multi-variable function. This leads us to the
+family of gradient descent methods. The latter are the working horses
+of basically all modern machine learning algorithms.
-The most common situation we encounter when we apply logistic
-regression is that of two possible outcomes, normally denoted as a
-binary outcome, true or false, positive or negative, success or
-failure etc.
+We note also that many of the topics discussed here on logistic
+regression are also commonly used in modern supervised Deep Learning
+models, as we will see later.
@@ -469,7 +462,7 @@ failure etc.
37
38
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs029.html b/doc/pub/week38/html/._week38-bs029.html
index d2e947ca1..1633e4aa3 100644
--- a/doc/pub/week38/html/._week38-bs029.html
+++ b/doc/pub/week38/html/._week38-bs029.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -421,25 +416,31 @@ MathJax.Hub.Config({
-
+
-Optimization and Deep learning
+Basics
-Logistic regression will also serve as our stepping stone towards
-neural network algorithms and supervised deep learning. For logistic
-learning, the minimization of the cost function leads to a non-linear
-equation in the parameters \( \boldsymbol{\beta} \). The optimization of the
-problem calls therefore for minimization algorithms. This forms the
-bottle neck of all machine learning algorithms, namely how to find
-reliable minima of a multi-variable function. This leads us to the
-family of gradient descent methods. The latter are the working horses
-of basically all modern machine learning algorithms.
+We consider the case where the dependent variables, also called the
+responses or the outcomes, \( y_i \) are discrete and only take values
+from \( k=0,\dots,K-1 \) (i.e. \( K \) classes).
-We note also that many of the topics discussed here on logistic
-regression are also commonly used in modern supervised Deep Learning
-models, as we will see later.
+The goal is to predict the
+output classes from the design matrix \( \boldsymbol{X}\in\mathbb{R}^{n\times p} \)
+made of \( n \) samples, each of which carries \( p \) features or predictors. The
+primary goal is to identify the classes to which new unseen samples
+belong.
+
+
+Let us specialize to the case of two classes only, with outputs
+\( y_i=0 \) and \( y_i=1 \). Our outcomes could represent the status of a
+credit card user that could default or not on her/his credit card
+debt. That is
+
+$$
+y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}.
+$$
@@ -467,7 +468,7 @@ models, as we will see later.
38
39
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs030.html b/doc/pub/week38/html/._week38-bs030.html
index 25f02bff3..ecae8af76 100644
--- a/doc/pub/week38/html/._week38-bs030.html
+++ b/doc/pub/week38/html/._week38-bs030.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -421,32 +416,29 @@ MathJax.Hub.Config({
-
+
-Basics
+Linear classifier
-We consider the case where the dependent variables, also called the
-responses or the outcomes, \( y_i \) are discrete and only take values
-from \( k=0,\dots,K-1 \) (i.e. \( K \) classes).
+Before moving to the logistic model, let us try to use our linear
+regression model to classify these two outcomes. We could for example
+fit a linear model to the default case if \( y_i > 0.5 \) and the no
+default case \( y_i \leq 0.5 \).
-The goal is to predict the
-output classes from the design matrix \( \boldsymbol{X}\in\mathbb{R}^{n\times p} \)
-made of \( n \) samples, each of which carries \( p \) features or predictors. The
-primary goal is to identify the classes to which new unseen samples
-belong.
-
-
-Let us specialize to the case of two classes only, with outputs
-\( y_i=0 \) and \( y_i=1 \). Our outcomes could represent the status of a
-credit card user that could default or not on her/his credit card
-debt. That is
-
+We would then have our
+weighted linear combination, namely
$$
-y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}.
+\begin{equation}
+\boldsymbol{y} = \boldsymbol{X}^T\boldsymbol{\beta} + \boldsymbol{\epsilon},
+\tag{13}
+\end{equation}
$$
+where \( \boldsymbol{y} \) is a vector representing the possible outcomes, \( \boldsymbol{X} \) is our
+\( n\times p \) design matrix and \( \boldsymbol{\beta} \) represents our estimators/predictors.
+
@@ -473,7 +465,7 @@ $$
39
40
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs031.html b/doc/pub/week38/html/._week38-bs031.html
index 5e191feaf..806addd4a 100644
--- a/doc/pub/week38/html/._week38-bs031.html
+++ b/doc/pub/week38/html/._week38-bs031.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,26 +418,24 @@ MathJax.Hub.Config({
-Linear classifier
+Some selected properties
-Before moving to the logistic model, let us try to use our linear
-regression model to classify these two outcomes. We could for example
-fit a linear model to the default case if \( y_i > 0.5 \) and the no
-default case \( y_i \leq 0.5 \).
+The main problem with our function is that it takes values on the
+entire real axis. In the case of logistic regression, however, the
+labels \( y_i \) are discrete variables. A typical example is the credit
+card data discussed below here, where we can set the state of
+defaulting the debt to \( y_i=1 \) and not to \( y_i=0 \) for one the persons
+in the data set (see the full example below).
-We would then have our
-weighted linear combination, namely
-$$
-\begin{equation}
-\boldsymbol{y} = \boldsymbol{X}^T\boldsymbol{\beta} + \boldsymbol{\epsilon},
-\tag{13}
-\end{equation}
-$$
-
-where \( \boldsymbol{y} \) is a vector representing the possible outcomes, \( \boldsymbol{X} \) is our
-\( n\times p \) design matrix and \( \boldsymbol{\beta} \) represents our estimators/predictors.
+One simple way to get a discrete output is to have sign
+functions that map the output of a linear regressor to values \( \{0,1\} \),
+\( f(s_i)=sign(s_i)=1 \) if \( s_i\ge 0 \) and 0 if otherwise.
+We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron" model in the machine learning
+literature. This model is extremely simple. However, in many cases it is more
+favorable to use a ``soft" classifier that outputs
+the probability of a given category. This leads us to the logistic function.
@@ -470,7 +463,7 @@ where \( \boldsymbol{y} \) is a vector representing the possible outcomes, \( \b
40
41
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs032.html b/doc/pub/week38/html/._week38-bs032.html
index 784425453..d4bbd8380 100644
--- a/doc/pub/week38/html/._week38-bs032.html
+++ b/doc/pub/week38/html/._week38-bs032.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,25 +418,69 @@ MathJax.Hub.Config({
-Some selected properties
+Simple example
-The main problem with our function is that it takes values on the
-entire real axis. In the case of logistic regression, however, the
-labels \( y_i \) are discrete variables. A typical example is the credit
-card data discussed below here, where we can set the state of
-defaulting the debt to \( y_i=1 \) and not to \( y_i=0 \) for one the persons
-in the data set (see the full example below).
+The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful.
-One simple way to get a discrete output is to have sign
-functions that map the output of a linear regressor to values \( \{0,1\} \),
-\( f(s_i)=sign(s_i)=1 \) if \( s_i\ge 0 \) and 0 if otherwise.
-We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron" model in the machine learning
-literature. This model is extremely simple. However, in many cases it is more
-favorable to use a ``soft" classifier that outputs
-the probability of a given category. This leads us to the logistic function.
+
+
# Common imports
+import os
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.linear_model import LinearRegression, Ridge, Lasso
+from sklearn.model_selection import train_test_split
+from sklearn.utils import resample
+from sklearn.metrics import mean_squared_error
+from IPython.display import display
+from pylab import plt, mpl
+plt.style.use('seaborn')
+mpl.rcParams['font.family'] = 'serif'
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+ os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+ os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+ os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+ return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+ return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+ plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("chddata.csv"),'r')
+
+# Read the chd data as csv file and organize the data into arrays with age group, age, and chd
+chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))
+chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']
+output = chd['CHD']
+age = chd['Age']
+agegroup = chd['Agegroup']
+numberID = chd['ID']
+display(chd)
+
+plt.scatter(age, output, marker='o')
+plt.axis([18,70.0,-0.1, 1.2])
+plt.xlabel(r'Age')
+plt.ylabel(r'CHD')
+plt.title(r'Age distribution and Coronary heart disease')
+plt.show()
+
@@ -468,7 +507,7 @@ the probability of a given category. This leads us to the logistic function.
41
42
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs033.html b/doc/pub/week38/html/._week38-bs033.html
index a22a99d32..aae61091f 100644
--- a/doc/pub/week38/html/._week38-bs033.html
+++ b/doc/pub/week38/html/._week38-bs033.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,69 +418,42 @@ MathJax.Hub.Config({
-Simple example
+Plotting the mean value for each group
-The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful.
+What we could attempt however is to plot the mean value for each group.
-
# Common imports
-import os
-import numpy as np
-import pandas as pd
-import matplotlib.pyplot as plt
-from sklearn.linear_model import LinearRegression, Ridge, Lasso
-from sklearn.model_selection import train_test_split
-from sklearn.utils import resample
-from sklearn.metrics import mean_squared_error
-from IPython.display import display
-from pylab import plt, mpl
-plt.style.use('seaborn')
-mpl.rcParams['font.family'] = 'serif'
-
-# Where to save the figures and data files
-PROJECT_ROOT_DIR = "Results"
-FIGURE_ID = "Results/FigureFiles"
-DATA_ID = "DataFiles/"
-
-if not os.path.exists(PROJECT_ROOT_DIR):
- os.mkdir(PROJECT_ROOT_DIR)
-
-if not os.path.exists(FIGURE_ID):
- os.makedirs(FIGURE_ID)
-
-if not os.path.exists(DATA_ID):
- os.makedirs(DATA_ID)
-
-def image_path(fig_id):
- return os.path.join(FIGURE_ID, fig_id)
-
-def data_path(dat_id):
- return os.path.join(DATA_ID, dat_id)
-
-def save_fig(fig_id):
- plt.savefig(image_path(fig_id) + ".png", format='png')
-
-infile = open(data_path("chddata.csv"),'r')
-
-# Read the chd data as csv file and organize the data into arrays with age group, age, and chd
-chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))
-chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']
-output = chd['CHD']
-age = chd['Age']
-agegroup = chd['Agegroup']
-numberID = chd['ID']
-display(chd)
-
-plt.scatter(age, output, marker='o')
-plt.axis([18,70.0,-0.1, 1.2])
-plt.xlabel(r'Age')
-plt.ylabel(r'CHD')
-plt.title(r'Age distribution and Coronary heart disease')
+agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])
+group = np.array([1, 2, 3, 4, 5, 6, 7, 8])
+plt.plot(group, agegroupmean, "r-")
+plt.axis([0,9,0, 1.0])
+plt.xlabel(r'Age group')
+plt.ylabel(r'CHD mean values')
+plt.title(r'Mean values for each age group')
plt.show()
+
+We are now trying to find a function \( f(y\vert x) \), that is a function which gives us an expected value for the output \( y \) with a given input \( x \).
+In standard linear regression with a linear dependence on \( x \), we would write this in terms of our model
+$$
+f(y_i\vert x_i)=\beta_0+\beta_1 x_i.
+$$
+
+
+This expression implies however that \( f(y_i\vert x_i) \) could take any
+value from minus infinity to plus infinity. If we however let
+\( f(y\vert y) \) be represented by the mean value, the above example
+shows us that we can constrain the function to take values between
+zero and one, that is we have \( 0 \le f(y_i\vert x_i) \le 1 \). Looking
+at our last curve we see also that it has an S-shaped form. This leads
+us to a very popular model for the function \( f \), namely the so-called
+Sigmoid function or logistic model. We will consider this function as
+representing the probability for finding a value of \( y_i \) with a given
+\( x_i \).
+
@@ -512,7 +480,7 @@ plt.show()
42
43
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs034.html b/doc/pub/week38/html/._week38-bs034.html
index b4df67c5e..850014b15 100644
--- a/doc/pub/week38/html/._week38-bs034.html
+++ b/doc/pub/week38/html/._week38-bs034.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,41 +418,25 @@ MathJax.Hub.Config({
-Plotting the mean value for each group
+The logistic function
-What we could attempt however is to plot the mean value for each group.
-
-
-
-
-
agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])
-group = np.array([1, 2, 3, 4, 5, 6, 7, 8])
-plt.plot(group, agegroupmean, "r-")
-plt.axis([0,9,0, 1.0])
-plt.xlabel(r'Age group')
-plt.ylabel(r'CHD mean values')
-plt.title(r'Mean values for each age group')
-plt.show()
-
-
-We are now trying to find a function \( f(y\vert x) \), that is a function which gives us an expected value for the output \( y \) with a given input \( x \).
-In standard linear regression with a linear dependence on \( x \), we would write this in terms of our model
+Another widely studied model, is the so-called
+perceptron model, which is an example of a "hard classification" model. We
+will encounter this model when we discuss neural networks as
+well. Each datapoint is deterministically assigned to a category (i.e
+\( y_i=0 \) or \( y_i=1 \)). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a "soft"
+classifier that outputs the probability of a given category rather
+than a single value. For example, given \( x_i \), the classifier
+outputs the probability of being in a category \( k \). Logistic regression
+is the most common example of a so-called soft classifier. In logistic
+regression, the probability that a data point \( x_i \)
+belongs to a category \( y_i=\{0,1\} \) is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event,
$$
-f(y_i\vert x_i)=\beta_0+\beta_1 x_i.
+p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}.
$$
-
-This expression implies however that \( f(y_i\vert x_i) \) could take any
-value from minus infinity to plus infinity. If we however let
-\( f(y\vert y) \) be represented by the mean value, the above example
-shows us that we can constrain the function to take values between
-zero and one, that is we have \( 0 \le f(y_i\vert x_i) \le 1 \). Looking
-at our last curve we see also that it has an S-shaped form. This leads
-us to a very popular model for the function \( f \), namely the so-called
-Sigmoid function or logistic model. We will consider this function as
-representing the probability for finding a value of \( y_i \) with a given
-\( x_i \).
+Note that \( 1-p(t)= p(-t) \).
@@ -485,7 +464,7 @@ representing the probability for finding a value of \( y_i \) with a given
43
44
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs035.html b/doc/pub/week38/html/._week38-bs035.html
index e39c002e8..4acbc04d7 100644
--- a/doc/pub/week38/html/._week38-bs035.html
+++ b/doc/pub/week38/html/._week38-bs035.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,26 +418,69 @@ MathJax.Hub.Config({
-The logistic function
+Examples of likelihood functions used in logistic regression and nueral networks
-Another widely studied model, is the so-called
-perceptron model, which is an example of a "hard classification" model. We
-will encounter this model when we discuss neural networks as
-well. Each datapoint is deterministically assigned to a category (i.e
-\( y_i=0 \) or \( y_i=1 \)). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a "soft"
-classifier that outputs the probability of a given category rather
-than a single value. For example, given \( x_i \), the classifier
-outputs the probability of being in a category \( k \). Logistic regression
-is the most common example of a so-called soft classifier. In logistic
-regression, the probability that a data point \( x_i \)
-belongs to a category \( y_i=\{0,1\} \) is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event,
-$$
-p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}.
-$$
+The following code plots the logistic function, the step function and other functions we will encounter from here and on.
-Note that \( 1-p(t)= p(-t) \).
+
+
+
"""The sigmoid function (or the logistic curve) is a
+function that takes any real number, z, and outputs a number (0,1).
+It is useful in neural networks for assigning weights on a relative scale.
+The value z is the weighted sum of parameters involved in the learning algorithm."""
+
+import numpy
+import matplotlib.pyplot as plt
+import math as mt
+
+z = numpy.arange(-5, 5, .1)
+sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))
+sigma = sigma_fn(z)
+
+fig = plt.figure()
+ax = fig.add_subplot(111)
+ax.plot(z, sigma)
+ax.set_ylim([-0.1, 1.1])
+ax.set_xlim([-5,5])
+ax.grid(True)
+ax.set_xlabel('z')
+ax.set_title('sigmoid function')
+
+plt.show()
+
+"""Step Function"""
+z = numpy.arange(-5, 5, .02)
+step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)
+step = step_fn(z)
+
+fig = plt.figure()
+ax = fig.add_subplot(111)
+ax.plot(z, step)
+ax.set_ylim([-0.5, 1.5])
+ax.set_xlim([-5,5])
+ax.grid(True)
+ax.set_xlabel('z')
+ax.set_title('step function')
+
+plt.show()
+
+"""tanh Function"""
+z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)
+t = numpy.tanh(z)
+
+fig = plt.figure()
+ax = fig.add_subplot(111)
+ax.plot(z, t)
+ax.set_ylim([-1.0, 1.0])
+ax.set_xlim([-2*mt.pi,2*mt.pi])
+ax.grid(True)
+ax.set_xlabel('z')
+ax.set_title('tanh function')
+
+plt.show()
+
@@ -469,7 +507,7 @@ Note that \( 1-p(t)= p(-t) \).
44
45
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs036.html b/doc/pub/week38/html/._week38-bs036.html
index a53aa23bb..e6c59bd28 100644
--- a/doc/pub/week38/html/._week38-bs036.html
+++ b/doc/pub/week38/html/._week38-bs036.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,69 +418,25 @@ MathJax.Hub.Config({
-Examples of likelihood functions used in logistic regression and nueral networks
+Two parameters
-The following code plots the logistic function, the step function and other functions we will encounter from here and on.
+We assume now that we have two classes with \( y_i \) either \( 0 \) or \( 1 \). Furthermore we assume also that we have only two parameters \( \beta \) in our fitting of the Sigmoid function, that is we define probabilities
+$$
+\begin{align*}
+p(y_i=1|x_i,\boldsymbol{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\
+p(y_i=0|x_i,\boldsymbol{\beta}) &= 1 - p(y_i=1|x_i,\boldsymbol{\beta}),
+\end{align*}
+$$
+
+where \( \boldsymbol{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).
+Note that we used
+$$
+p(y_i=0\vert x_i, \boldsymbol{\beta}) = 1-p(y_i=1\vert x_i, \boldsymbol{\beta}).
+$$
-
-
"""The sigmoid function (or the logistic curve) is a
-function that takes any real number, z, and outputs a number (0,1).
-It is useful in neural networks for assigning weights on a relative scale.
-The value z is the weighted sum of parameters involved in the learning algorithm."""
-
-import numpy
-import matplotlib.pyplot as plt
-import math as mt
-
-z = numpy.arange(-5, 5, .1)
-sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))
-sigma = sigma_fn(z)
-
-fig = plt.figure()
-ax = fig.add_subplot(111)
-ax.plot(z, sigma)
-ax.set_ylim([-0.1, 1.1])
-ax.set_xlim([-5,5])
-ax.grid(True)
-ax.set_xlabel('z')
-ax.set_title('sigmoid function')
-
-plt.show()
-
-"""Step Function"""
-z = numpy.arange(-5, 5, .02)
-step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)
-step = step_fn(z)
-
-fig = plt.figure()
-ax = fig.add_subplot(111)
-ax.plot(z, step)
-ax.set_ylim([-0.5, 1.5])
-ax.set_xlim([-5,5])
-ax.grid(True)
-ax.set_xlabel('z')
-ax.set_title('step function')
-
-plt.show()
-
-"""tanh Function"""
-z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)
-t = numpy.tanh(z)
-
-fig = plt.figure()
-ax = fig.add_subplot(111)
-ax.plot(z, t)
-ax.set_ylim([-1.0, 1.0])
-ax.set_xlim([-2*mt.pi,2*mt.pi])
-ax.grid(True)
-ax.set_xlabel('z')
-ax.set_title('tanh function')
-
-plt.show()
-
@@ -512,7 +463,7 @@ plt.show()
45
46
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs037.html b/doc/pub/week38/html/._week38-bs037.html
index 0a66ee898..63291841f 100644
--- a/doc/pub/week38/html/._week38-bs037.html
+++ b/doc/pub/week38/html/._week38-bs037.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -421,25 +416,26 @@ MathJax.Hub.Config({
-
+
-Two parameters
+Maximum likelihood
-We assume now that we have two classes with \( y_i \) either \( 0 \) or \( 1 \). Furthermore we assume also that we have only two parameters \( \beta \) in our fitting of the Sigmoid function, that is we define probabilities
+In order to define the total likelihood for all possible outcomes from a
+dataset \( \mathcal{D}=\{(y_i,x_i)\} \), with the binary labels
+\( y_i\in\{0,1\} \) and where the data points are drawn independently, we use the so-called Maximum Likelihood Estimation (MLE) principle.
+We aim thus at maximizing
+the probability of seeing the observed data. We can then approximate the
+likelihood in terms of the product of the individual probabilities of a specific outcome \( y_i \), that is
$$
\begin{align*}
-p(y_i=1|x_i,\boldsymbol{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\
-p(y_i=0|x_i,\boldsymbol{\beta}) &= 1 - p(y_i=1|x_i,\boldsymbol{\beta}),
+P(\mathcal{D}|\boldsymbol{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\boldsymbol{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\boldsymbol{\beta}))\right]^{1-y_i}\nonumber \\
\end{align*}
$$
-where \( \boldsymbol{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).
-
-
-Note that we used
+from which we obtain the log-likelihood and our cost/loss function
$$
-p(y_i=0\vert x_i, \boldsymbol{\beta}) = 1-p(y_i=1\vert x_i, \boldsymbol{\beta}).
+\mathcal{C}(\boldsymbol{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\boldsymbol{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\boldsymbol{\beta}))\right]\right).
$$
@@ -468,7 +464,7 @@ $$
46
47
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs038.html b/doc/pub/week38/html/._week38-bs038.html
index fedb97b7a..58c4b249a 100644
--- a/doc/pub/week38/html/._week38-bs038.html
+++ b/doc/pub/week38/html/._week38-bs038.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -421,28 +416,26 @@ MathJax.Hub.Config({
-
+
-Maximum likelihood
+The cost function rewritten
-In order to define the total likelihood for all possible outcomes from a
-dataset \( \mathcal{D}=\{(y_i,x_i)\} \), with the binary labels
-\( y_i\in\{0,1\} \) and where the data points are drawn independently, we use the so-called Maximum Likelihood Estimation (MLE) principle.
-We aim thus at maximizing
-the probability of seeing the observed data. We can then approximate the
-likelihood in terms of the product of the individual probabilities of a specific outcome \( y_i \), that is
+Reordering the logarithms, we can rewrite the cost/loss function as
$$
-\begin{align*}
-P(\mathcal{D}|\boldsymbol{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\boldsymbol{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\boldsymbol{\beta}))\right]^{1-y_i}\nonumber \\
-\end{align*}
+\mathcal{C}(\boldsymbol{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
$$
-from which we obtain the log-likelihood and our cost/loss function
+
+The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to \( \beta \).
+Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that
$$
-\mathcal{C}(\boldsymbol{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\boldsymbol{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\boldsymbol{\beta}))\right]\right).
+\mathcal{C}(\boldsymbol{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
$$
+This equation is known in statistics as the cross entropy. Finally, we note that just as in linear regression,
+in practice we often supplement the cross-entropy with additional regularization terms, usually \( L_1 \) and \( L_2 \) regularization as we did for Ridge and Lasso regression.
+
@@ -469,7 +462,7 @@ $$
47
48
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs039.html b/doc/pub/week38/html/._week38-bs039.html
index 7d67584f0..e177ad1ab 100644
--- a/doc/pub/week38/html/._week38-bs039.html
+++ b/doc/pub/week38/html/._week38-bs039.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,23 +418,24 @@ MathJax.Hub.Config({
-The cost function rewritten
+Minimizing the cross entropy
-Reordering the logarithms, we can rewrite the cost/loss function as
-$$
-\mathcal{C}(\boldsymbol{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
-$$
+The cross entropy is a convex function of the weights \( \boldsymbol{\beta} \) and,
+therefore, any local minimizer is a global minimizer.
-The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to \( \beta \).
-Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that
+Minimizing this
+cost function with respect to the two parameters \( \beta_0 \) and \( \beta_1 \) we obtain
+
$$
-\mathcal{C}(\boldsymbol{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
+\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right),
$$
-This equation is known in statistics as the cross entropy. Finally, we note that just as in linear regression,
-in practice we often supplement the cross-entropy with additional regularization terms, usually \( L_1 \) and \( L_2 \) regularization as we did for Ridge and Lasso regression.
+and
+$$
+\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right).
+$$
@@ -467,7 +463,7 @@ in practice we often supplement the cross-entropy with additional regularization
48
49
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs040.html b/doc/pub/week38/html/._week38-bs040.html
index 0f0d15c2d..6010937b7 100644
--- a/doc/pub/week38/html/._week38-bs040.html
+++ b/doc/pub/week38/html/._week38-bs040.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,23 +418,24 @@ MathJax.Hub.Config({
-Minimizing the cross entropy
+A more compact expression
-The cross entropy is a convex function of the weights \( \boldsymbol{\beta} \) and,
-therefore, any local minimizer is a global minimizer.
+Let us now define a vector \( \boldsymbol{y} \) with \( n \) elements \( y_i \), an
+\( n\times p \) matrix \( \boldsymbol{X} \) which contains the \( x_i \) values and a
+vector \( \boldsymbol{p} \) of fitted probabilities \( p(y_i\vert x_i,\boldsymbol{\beta}) \). We can rewrite in a more compact form the first
+derivative of cost function as
+
+$$
+\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{p}\right).
+$$
-Minimizing this
-cost function with respect to the two parameters \( \beta_0 \) and \( \beta_1 \) we obtain
+If we in addition define a diagonal matrix \( \boldsymbol{W} \) with elements
+\( p(y_i\vert x_i,\boldsymbol{\beta})(1-p(y_i\vert x_i,\boldsymbol{\beta}) \), we can obtain a compact expression of the second derivative as
$$
-\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right),
-$$
-
-and
-$$
-\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right).
+\frac{\partial^2 \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}\partial \boldsymbol{\beta}^T} = \boldsymbol{X}^T\boldsymbol{W}\boldsymbol{X}.
$$
@@ -468,7 +464,7 @@ $$
49
50
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs041.html b/doc/pub/week38/html/._week38-bs041.html
index ba527bc21..1024a0957 100644
--- a/doc/pub/week38/html/._week38-bs041.html
+++ b/doc/pub/week38/html/._week38-bs041.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,24 +418,17 @@ MathJax.Hub.Config({
-A more compact expression
+Extending to more predictors
-Let us now define a vector \( \boldsymbol{y} \) with \( n \) elements \( y_i \), an
-\( n\times p \) matrix \( \boldsymbol{X} \) which contains the \( x_i \) values and a
-vector \( \boldsymbol{p} \) of fitted probabilities \( p(y_i\vert x_i,\boldsymbol{\beta}) \). We can rewrite in a more compact form the first
-derivative of cost function as
-
+Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with \( p \) predictors
$$
-\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{p}\right).
+\log{ \frac{p(\boldsymbol{\beta}\boldsymbol{x})}{1-p(\boldsymbol{\beta}\boldsymbol{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p.
$$
-
-If we in addition define a diagonal matrix \( \boldsymbol{W} \) with elements
-\( p(y_i\vert x_i,\boldsymbol{\beta})(1-p(y_i\vert x_i,\boldsymbol{\beta}) \), we can obtain a compact expression of the second derivative as
-
+Here we defined \( \boldsymbol{x}=[1,x_1,x_2,\dots,x_p] \) and \( \boldsymbol{\beta}=[\beta_0, \beta_1, \dots, \beta_p] \) leading to
$$
-\frac{\partial^2 \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}\partial \boldsymbol{\beta}^T} = \boldsymbol{X}^T\boldsymbol{W}\boldsymbol{X}.
+p(\boldsymbol{\beta}\boldsymbol{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}.
$$
@@ -469,7 +457,7 @@ $$
50
51
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/._week38-bs042.html b/doc/pub/week38/html/._week38-bs042.html
index df0ea302a..6e8f03803 100644
--- a/doc/pub/week38/html/._week38-bs042.html
+++ b/doc/pub/week38/html/._week38-bs042.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -423,19 +418,31 @@ MathJax.Hub.Config({
-Extending to more predictors
+Including more classes
-Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with \( p \) predictors
+Till now we have mainly focused on two classes, the so-called binary
+system. Suppose we wish to extend to \( K \) classes. Let us for the sake
+of simplicity assume we have only two predictors. We have then following model
+
$$
-\log{ \frac{p(\boldsymbol{\beta}\boldsymbol{x})}{1-p(\boldsymbol{\beta}\boldsymbol{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p.
+\log{\frac{p(C=1\vert x)}{p(K\vert x)}} = \beta_{10}+\beta_{11}x_1,
$$
-Here we defined \( \boldsymbol{x}=[1,x_1,x_2,\dots,x_p] \) and \( \boldsymbol{\beta}=[\beta_0, \beta_1, \dots, \beta_p] \) leading to
+and
$$
-p(\boldsymbol{\beta}\boldsymbol{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}.
+\log{\frac{p(C=2\vert x)}{p(K\vert x)}} = \beta_{20}+\beta_{21}x_1,
$$
+and so on till the class \( C=K-1 \) class
+$$
+\log{\frac{p(C=K-1\vert x)}{p(K\vert x)}} = \beta_{(K-1)0}+\beta_{(K-1)1}x_1,
+$$
+
+
+and the model is specified in term of \( K-1 \) so-called log-odds or
+logit transformations.
+
@@ -462,7 +469,7 @@ $$
51
52
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/week38-bs.html b/doc/pub/week38/html/week38-bs.html
index 406c60fa3..d5042016a 100644
--- a/doc/pub/week38/html/week38-bs.html
+++ b/doc/pub/week38/html/week38-bs.html
@@ -81,10 +81,6 @@ Automatically generated HTML file from DocOnce source
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -331,83 +327,82 @@ MathJax.Hub.Config({
Further Manipulations
Wrapping it up
Linear Regression code, Intercept handling first
- Another straight Line, the Intercept and Scaling again
- Code Examples
- Taking out the mean
- More complicated Example: The Ising model
- Reformulating the problem to suit regression
- Linear regression
- Singular Value decomposition
- The one-dimensional Ising model
- Ridge regression
- LASSO regression
- Performance as function of the regularization parameter
- Finding the optimal value of \( \lambda \)
- Logistic Regression
- Classification problems
- Optimization and Deep learning
- Basics
- Linear classifier
- Some selected properties
- Simple example
- Plotting the mean value for each group
- The logistic function
- Examples of likelihood functions used in logistic regression and nueral networks
- Two parameters
- Maximum likelihood
- The cost function rewritten
- Minimizing the cross entropy
- A more compact expression
- Extending to more predictors
- Including more classes
- More classes
- Friday September 24
- Wisconsin Cancer Data
- Using the correlation matrix
- Discussing the correlation data
- Other measures in classification studies: Cancer Data again
- Optimization, the central part of any Machine Learning algortithm
- Revisiting our Logistic Regression case
- The equations to solve
- Solving using Newton-Raphson's method
- Brief reminder on Newton-Raphson's method
- The equations
- Simple geometric interpretation
- Extending to more than one variable
- Steepest descent
- More on Steepest descent
- The ideal
- The sensitiveness of the gradient descent
- Convex functions
- Convex function
- Conditions on convex functions
- More on convex functions
- Some simple problems
- Friday September 25
- Standard steepest descent
- Gradient method
- Steepest descent method
- Steepest descent method
- Final expressions
- Steepest descent example
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method and iterations
- Conjugate gradient method
- Conjugate gradient method
- Conjugate gradient method
- Revisiting some of our first Linear Regression Encounters
- Gradient descent example
- The derivative of the cost/loss function
- The Hessian matrix
- Simple program
- Gradient Descent Example
- And a corresponding example using scikit-learn
- Gradient descent and Ridge
- Program example for gradient descent with Ridge Regression
- Using gradient descent methods, limitations
+ Code Examples
+ Taking out the mean
+ More complicated Example: The Ising model
+ Reformulating the problem to suit regression
+ Linear regression
+ Singular Value decomposition
+ The one-dimensional Ising model
+ Ridge regression
+ LASSO regression
+ Performance as function of the regularization parameter
+ Finding the optimal value of \( \lambda \)
+ Logistic Regression
+ Classification problems
+ Optimization and Deep learning
+ Basics
+ Linear classifier
+ Some selected properties
+ Simple example
+ Plotting the mean value for each group
+ The logistic function
+ Examples of likelihood functions used in logistic regression and nueral networks
+ Two parameters
+ Maximum likelihood
+ The cost function rewritten
+ Minimizing the cross entropy
+ A more compact expression
+ Extending to more predictors
+ Including more classes
+ More classes
+ Friday September 24
+ Wisconsin Cancer Data
+ Using the correlation matrix
+ Discussing the correlation data
+ Other measures in classification studies: Cancer Data again
+ Optimization, the central part of any Machine Learning algortithm
+ Revisiting our Logistic Regression case
+ The equations to solve
+ Solving using Newton-Raphson's method
+ Brief reminder on Newton-Raphson's method
+ The equations
+ Simple geometric interpretation
+ Extending to more than one variable
+ Steepest descent
+ More on Steepest descent
+ The ideal
+ The sensitiveness of the gradient descent
+ Convex functions
+ Convex function
+ Conditions on convex functions
+ More on convex functions
+ Some simple problems
+ Friday September 25
+ Standard steepest descent
+ Gradient method
+ Steepest descent method
+ Steepest descent method
+ Final expressions
+ Steepest descent example
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method and iterations
+ Conjugate gradient method
+ Conjugate gradient method
+ Conjugate gradient method
+ Revisiting some of our first Linear Regression Encounters
+ Gradient descent example
+ The derivative of the cost/loss function
+ The Hessian matrix
+ Simple program
+ Gradient Descent Example
+ And a corresponding example using scikit-learn
+ Gradient descent and Ridge
+ Program example for gradient descent with Ridge Regression
+ Using gradient descent methods, limitations
@@ -466,7 +461,7 @@ MathJax.Hub.Config({
9
10
...
- 92
+ 91
»
diff --git a/doc/pub/week38/html/week38-reveal.html b/doc/pub/week38/html/week38-reveal.html
index 758544922..da9264ad6 100644
--- a/doc/pub/week38/html/week38-reveal.html
+++ b/doc/pub/week38/html/week38-reveal.html
@@ -666,7 +666,7 @@ $$
-What does this mean for thus? And why do we insist on all this? Let us look at some examples.
+What does this mean? And why do we insist on all this? Let us look at some examples.
@@ -687,6 +687,10 @@ Note also that we do not split the data into training and test.
np.random.seed(2021)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
+
def fit_beta(X, y):
return np.linalg.pinv(X.T @ X) @ X.T @ y
@@ -714,6 +718,12 @@ clf = LinearRegression(fit_intercept=print(f"True beta: {true_beta}")
print(f"Fitted beta: {beta}")
print(f"Sklearn fitted beta: {clf.coef_}")
+ypredictOwn = X @ beta
+ypredictSKL = skl.predict(X)
+print(f"MSE with intercept column: {intercept}")
+print(MSE(y,ypredictOwn)
+print(f"MSE with intercept column from SKL: {intercept}")
+print(MSE(y,ypredictSKL)
plt.figure()
@@ -742,6 +752,12 @@ intercept = np.mean(y_offset - X_offset @ beta)
print(f"Fitted beta (wiothout intercept): {beta}")
print(f"Sklearn intercept: {clf.intercept_}")
print(f"Sklearn fitted beta (without intercept): {clf.coef_}")
+ypredictOwn = X @ beta
+ypredictSKL = skl.predict(X)
+print(f"MSE with Manual intercept: {intercept}")
+print(MSE(y,ypredictOwn)
+print(f"MSE with Sklearn intercept: {clf.intercept_}")
+print(MSE(y,ypredictSKL)
plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
plt.plot(x, clf.predict(X), "--", label="Sklearn (fit_intercept=True)")
@@ -750,65 +766,10 @@ plt.legend()
plt.show()
-
-
-
-
-Another straight Line, the Intercept and Scaling again
-
-As we said above, the intercept is the value of our output/target variable
-when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
-Let us now study a case where we again fit a straight line including the intercept in the design matrix
-using ordinary least squares. We are only interested in the MSE value here.
+The intercept is the value of our output/target variable
+when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
-
-
-
-
import numpy as np
-import pandas as pd
-import matplotlib.pyplot as plt
-from sklearn.model_selection import train_test_split
-from sklearn import linear_model
-
-def MSE(y_data,y_model):
- n = np.size(y_model)
- return np.sum((y_data-y_model)**2)/n
-
-def fit_beta(X, y):
- return np.linalg.pinv(X.T @ X) @ X.T @ y
-
-
-
-# A seed just to ensure that the random numbers are the same for every run.
-# Useful for eventual debugging.
-np.random.seed(3155)
-
-n = 100
-x = np.random.rand(n)
-y = 10.0+5*x
-
-
-Maxpolydegree = 2
-X = np.zeros((n,Maxpolydegree))
-
-for degree in range(Maxpolydegree):
- X[:,degree] = x**degree
-
-
-# We split the data in test and training data
-X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
-
-# Intercept is included in the design matrix
-# own code first
-OwnBeta = fit_beta(X_train, y_train)
-# Scikit-Learn
-skl = LinearRegression(fit_intercept=False).fit(X_train, y_train)
-ypredictOwn = X_test @ OwnBeta
-ypredictSKL = skl.predict(X_test)
-print(MSE(y_test,ypredictOwn)
-print(MSE(y_test,ypredictSKL)
-
Printing the MSE, we see first that both methods give the same
MSE. However, changing the value of the intercept gives a larger or
diff --git a/doc/pub/week38/html/week38-solarized.html b/doc/pub/week38/html/week38-solarized.html
index 38dcb81ba..c0fabbfa9 100644
--- a/doc/pub/week38/html/week38-solarized.html
+++ b/doc/pub/week38/html/week38-solarized.html
@@ -101,10 +101,6 @@ div { text-align: justify; text-justify: inter-word; }
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -804,7 +800,7 @@ $$
$$
-What does this mean for thus? And why do we insist on all this? Let us look at some examples.
+What does this mean? And why do we insist on all this? Let us look at some examples.
@@ -825,6 +821,10 @@ Note also that we do not split the data into training and test.
np.random.seed(2021)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
+
def fit_beta(X, y):
return np.linalg.pinv(X.T @ X) @ X.T @ y
@@ -852,6 +852,12 @@ clf = LinearRegression(fit_intercept=print(f"True beta: {true_beta}")
print(f"Fitted beta: {beta}")
print(f"Sklearn fitted beta: {clf.coef_}")
+ypredictOwn = X @ beta
+ypredictSKL = skl.predict(X)
+print(f"MSE with intercept column: {intercept}")
+print(MSE(y,ypredictOwn)
+print(f"MSE with intercept column from SKL: {intercept}")
+print(MSE(y,ypredictSKL)
plt.figure()
@@ -880,6 +886,12 @@ intercept = np.mean(y_offset - X_offset @ beta)
print(f"Fitted beta (wiothout intercept): {beta}")
print(f"Sklearn intercept: {clf.intercept_}")
print(f"Sklearn fitted beta (without intercept): {clf.coef_}")
+ypredictOwn = X @ beta
+ypredictSKL = skl.predict(X)
+print(f"MSE with Manual intercept: {intercept}")
+print(MSE(y,ypredictOwn)
+print(f"MSE with Sklearn intercept: {clf.intercept_}")
+print(MSE(y,ypredictSKL)
plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
plt.plot(x, clf.predict(X), "--", label="Sklearn (fit_intercept=True)")
@@ -889,63 +901,9 @@ plt.legend()
plt.show()
-
+The intercept is the value of our output/target variable
+when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
-
Another straight Line, the Intercept and Scaling again
-
-
-As we said above, the intercept is the value of our output/target variable
-when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
-Let us now study a case where we again fit a straight line including the intercept in the design matrix
-using ordinary least squares. We are only interested in the MSE value here.
-
-
-
-
-
import numpy as np
-import pandas as pd
-import matplotlib.pyplot as plt
-from sklearn.model_selection import train_test_split
-from sklearn import linear_model
-
-def MSE(y_data,y_model):
- n = np.size(y_model)
- return np.sum((y_data-y_model)**2)/n
-
-def fit_beta(X, y):
- return np.linalg.pinv(X.T @ X) @ X.T @ y
-
-
-
-# A seed just to ensure that the random numbers are the same for every run.
-# Useful for eventual debugging.
-np.random.seed(3155)
-
-n = 100
-x = np.random.rand(n)
-y = 10.0+5*x
-
-
-Maxpolydegree = 2
-X = np.zeros((n,Maxpolydegree))
-
-for degree in range(Maxpolydegree):
- X[:,degree] = x**degree
-
-
-# We split the data in test and training data
-X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
-
-# Intercept is included in the design matrix
-# own code first
-OwnBeta = fit_beta(X_train, y_train)
-# Scikit-Learn
-skl = LinearRegression(fit_intercept=False).fit(X_train, y_train)
-ypredictOwn = X_test @ OwnBeta
-ypredictSKL = skl.predict(X_test)
-print(MSE(y_test,ypredictOwn)
-print(MSE(y_test,ypredictSKL)
-
Printing the MSE, we see first that both methods give the same
MSE. However, changing the value of the intercept gives a larger or
diff --git a/doc/pub/week38/html/week38.html b/doc/pub/week38/html/week38.html
index b66923b1a..15a917992 100644
--- a/doc/pub/week38/html/week38.html
+++ b/doc/pub/week38/html/week38.html
@@ -106,10 +106,6 @@ div { text-align: justify; text-justify: inter-word; }
2,
None,
'linear-regression-code-intercept-handling-first'),
- ('Another straight Line, the Intercept and Scaling again',
- 2,
- None,
- 'another-straight-line-the-intercept-and-scaling-again'),
('Code Examples', 2, None, 'code-examples'),
('Taking out the mean', 2, None, 'taking-out-the-mean'),
('More complicated Example: The Ising model',
@@ -809,7 +805,7 @@ $$
$$
-What does this mean for thus? And why do we insist on all this? Let us look at some examples.
+What does this mean? And why do we insist on all this? Let us look at some examples.
@@ -830,6 +826,10 @@ Note also that we do not split the data into training and test.
np.random.seed(2021)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
+
def fit_beta(X, y):
return np.linalg.pinv(X.T @ X) @ X.T @ y
@@ -857,6 +857,12 @@ clf = LinearRegression(fit_interceptprint(f"True beta: {true_beta}")
print(f"Fitted beta: {beta}")
print(f"Sklearn fitted beta: {clf.coef_}")
+ypredictOwn = X @ beta
+ypredictSKL = skl.predict(X)
+print(f"MSE with intercept column: {intercept}")
+print(MSE(y,ypredictOwn)
+print(f"MSE with intercept column from SKL: {intercept}")
+print(MSE(y,ypredictSKL)
plt.figure()
@@ -885,6 +891,12 @@ intercept = np.
print(f"Fitted beta (wiothout intercept): {beta}")
print(f"Sklearn intercept: {clf.intercept_}")
print(f"Sklearn fitted beta (without intercept): {clf.coef_}")
+ypredictOwn = X @ beta
+ypredictSKL = skl.predict(X)
+print(f"MSE with Manual intercept: {intercept}")
+print(MSE(y,ypredictOwn)
+print(f"MSE with Sklearn intercept: {clf.intercept_}")
+print(MSE(y,ypredictSKL)
plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
plt.plot(x, clf.predict(X), "--", label="Sklearn (fit_intercept=True)")
@@ -894,63 +906,9 @@ plt.legend()
plt.show()
-
+The intercept is the value of our output/target variable
+when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
-
Another straight Line, the Intercept and Scaling again
-
-
-As we said above, the intercept is the value of our output/target variable
-when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
-Let us now study a case where we again fit a straight line including the intercept in the design matrix
-using ordinary least squares. We are only interested in the MSE value here.
-
-
-
-
-
import numpy as np
-import pandas as pd
-import matplotlib.pyplot as plt
-from sklearn.model_selection import train_test_split
-from sklearn import linear_model
-
-def MSE(y_data,y_model):
- n = np.size(y_model)
- return np.sum((y_data-y_model)**2)/n
-
-def fit_beta(X, y):
- return np.linalg.pinv(X.T @ X) @ X.T @ y
-
-
-
-# A seed just to ensure that the random numbers are the same for every run.
-# Useful for eventual debugging.
-np.random.seed(3155)
-
-n = 100
-x = np.random.rand(n)
-y = 10.0+5*x
-
-
-Maxpolydegree = 2
-X = np.zeros((n,Maxpolydegree))
-
-for degree in range(Maxpolydegree):
- X[:,degree] = x**degree
-
-
-# We split the data in test and training data
-X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
-
-# Intercept is included in the design matrix
-# own code first
-OwnBeta = fit_beta(X_train, y_train)
-# Scikit-Learn
-skl = LinearRegression(fit_intercept=False).fit(X_train, y_train)
-ypredictOwn = X_test @ OwnBeta
-ypredictSKL = skl.predict(X_test)
-print(MSE(y_test,ypredictOwn)
-print(MSE(y_test,ypredictSKL)
-
Printing the MSE, we see first that both methods give the same
MSE. However, changing the value of the intercept gives a larger or
diff --git a/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz b/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz
index 0d38a1c71..822c4fc4f 100644
Binary files a/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz and b/doc/pub/week38/ipynb/ipynb-week38-src.tar.gz differ
diff --git a/doc/pub/week38/ipynb/week38.ipynb b/doc/pub/week38/ipynb/week38.ipynb
index bf0cc69f9..d6ebe80a8 100644
--- a/doc/pub/week38/ipynb/week38.ipynb
+++ b/doc/pub/week38/ipynb/week38.ipynb
@@ -723,7 +723,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "What does this mean for thus? And why do we insist on all this? Let us look at some examples.\n",
+ "What does this mean? And why do we insist on all this? Let us look at some examples.\n",
"\n",
"\n",
"\n",
@@ -750,6 +750,10 @@
"\n",
"np.random.seed(2021)\n",
"\n",
+ "def MSE(y_data,y_model):\n",
+ " n = np.size(y_model)\n",
+ " return np.sum((y_data-y_model)**2)/n\n",
+ "\n",
"\n",
"def fit_beta(X, y):\n",
" return np.linalg.pinv(X.T @ X) @ X.T @ y\n",
@@ -777,6 +781,12 @@
"print(f\"True beta: {true_beta}\")\n",
"print(f\"Fitted beta: {beta}\")\n",
"print(f\"Sklearn fitted beta: {clf.coef_}\")\n",
+ "ypredictOwn = X @ beta\n",
+ "ypredictSKL = skl.predict(X)\n",
+ "print(f\"MSE with intercept column: {intercept}\")\n",
+ "print(MSE(y,ypredictOwn)\n",
+ "print(f\"MSE with intercept column from SKL: {intercept}\")\n",
+ "print(MSE(y,ypredictSKL)\n",
"\n",
"\n",
"plt.figure()\n",
@@ -805,6 +815,12 @@
"print(f\"Fitted beta (wiothout intercept): {beta}\")\n",
"print(f\"Sklearn intercept: {clf.intercept_}\")\n",
"print(f\"Sklearn fitted beta (without intercept): {clf.coef_}\")\n",
+ "ypredictOwn = X @ beta\n",
+ "ypredictSKL = skl.predict(X)\n",
+ "print(f\"MSE with Manual intercept: {intercept}\")\n",
+ "print(MSE(y,ypredictOwn)\n",
+ "print(f\"MSE with Sklearn intercept: {clf.intercept_}\")\n",
+ "print(MSE(y,ypredictSKL)\n",
"\n",
"plt.plot(x, X @ beta + intercept, \"--\", label=\"Fit (manual intercept)\")\n",
"plt.plot(x, clf.predict(X), \"--\", label=\"Sklearn (fit_intercept=True)\")\n",
@@ -818,72 +834,9 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "## Another straight Line, the Intercept and Scaling again\n",
- "\n",
- "As we said above, the intercept is the value of our output/target variable\n",
+ "The intercept is the value of our output/target variable\n",
"when all our features are zero and our function crosses the $y$-axis (for a one-dimensional case). \n",
- "Let us now study a case where we again fit a straight line including the intercept in the design matrix\n",
- "using ordinary least squares. We are only interested in the MSE value here."
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "metadata": {
- "collapsed": false,
- "editable": true
- },
- "outputs": [],
- "source": [
- "import numpy as np\n",
- "import pandas as pd\n",
- "import matplotlib.pyplot as plt\n",
- "from sklearn.model_selection import train_test_split\n",
- "from sklearn import linear_model\n",
"\n",
- "def MSE(y_data,y_model):\n",
- " n = np.size(y_model)\n",
- " return np.sum((y_data-y_model)**2)/n\n",
- "\n",
- "def fit_beta(X, y):\n",
- " return np.linalg.pinv(X.T @ X) @ X.T @ y\n",
- "\n",
- "\n",
- "\n",
- "# A seed just to ensure that the random numbers are the same for every run.\n",
- "# Useful for eventual debugging.\n",
- "np.random.seed(3155)\n",
- "\n",
- "n = 100\n",
- "x = np.random.rand(n)\n",
- "y = 10.0+5*x\n",
- "\n",
- "\n",
- "Maxpolydegree = 2\n",
- "X = np.zeros((n,Maxpolydegree))\n",
- "\n",
- "for degree in range(Maxpolydegree):\n",
- " X[:,degree] = x**degree\n",
- "\n",
- "\n",
- "# We split the data in test and training data\n",
- "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n",
- "\n",
- "# Intercept is included in the design matrix\n",
- "# own code first\n",
- "OwnBeta = fit_beta(X_train, y_train)\n",
- "# Scikit-Learn\n",
- "skl = LinearRegression(fit_intercept=False).fit(X_train, y_train)\n",
- "ypredictOwn = X_test @ OwnBeta\n",
- "ypredictSKL = skl.predict(X_test)\n",
- "print(MSE(y_test,ypredictOwn)\n",
- "print(MSE(y_test,ypredictSKL)"
- ]
- },
- {
- "cell_type": "markdown",
- "metadata": {},
- "source": [
"Printing the MSE, we see first that both methods give the same\n",
"MSE. However, changing the value of the intercept gives a larger or\n",
"smaller MSE, meaning that the MSE is penalized by the value of the\n",
@@ -892,7 +845,6 @@
"independent of the specific value of the intercept.\n",
"\n",
"\n",
- "\n",
"## Code Examples\n",
"\n",
"Armed with this wisdom, we attempt first simply set the intercept eqault to **False** in our implementation of Ridge regression for yet another vanilla data set."
diff --git a/doc/src/week38/week38.do.txt b/doc/src/week38/week38.do.txt
index b92c1ad9a..5ef70fba8 100644
--- a/doc/src/week38/week38.do.txt
+++ b/doc/src/week38/week38.do.txt
@@ -448,7 +448,7 @@ For Ridge regression we need to add $\lambda \boldsymbol{\beta}^T\boldsymbol{\be
\]
!et
-What does this mean for thus? And why do we insist on all this? Let us look at some examples.
+What does this mean? And why do we insist on all this? Let us look at some examples.
@@ -466,6 +466,10 @@ from sklearn.linear_model import LinearRegression
np.random.seed(2021)
+def MSE(y_data,y_model):
+ n = np.size(y_model)
+ return np.sum((y_data-y_model)**2)/n
+
def fit_beta(X, y):
return np.linalg.pinv(X.T @ X) @ X.T @ y
@@ -493,6 +497,12 @@ clf = LinearRegression(fit_intercept=False).fit(X, y)
print(f"True beta: {true_beta}")
print(f"Fitted beta: {beta}")
print(f"Sklearn fitted beta: {clf.coef_}")
+ypredictOwn = X @ beta
+ypredictSKL = skl.predict(X)
+print(f"MSE with intercept column: {intercept}")
+print(MSE(y,ypredictOwn)
+print(f"MSE with intercept column from SKL: {intercept}")
+print(MSE(y,ypredictSKL)
plt.figure()
@@ -521,6 +531,12 @@ print(f"Manual intercept: {intercept}")
print(f"Fitted beta (wiothout intercept): {beta}")
print(f"Sklearn intercept: {clf.intercept_}")
print(f"Sklearn fitted beta (without intercept): {clf.coef_}")
+ypredictOwn = X @ beta
+ypredictSKL = skl.predict(X)
+print(f"MSE with Manual intercept: {intercept}")
+print(MSE(y,ypredictOwn)
+print(f"MSE with Sklearn intercept: {clf.intercept_}")
+print(MSE(y,ypredictSKL)
plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
plt.plot(x, clf.predict(X), "--", label="Sklearn (fit_intercept=True)")
@@ -531,63 +547,8 @@ plt.show()
!ec
-
-
-
-!split
-===== Another straight Line, the Intercept and Scaling again =====
-
-As we said above, the intercept is the value of our output/target variable
+The intercept is the value of our output/target variable
when all our features are zero and our function crosses the $y$-axis (for a one-dimensional case).
-Let us now study a case where we again fit a straight line including the intercept in the design matrix
-using ordinary least squares. We are only interested in the MSE value here.
-
-!bc pycod
-import numpy as np
-import pandas as pd
-import matplotlib.pyplot as plt
-from sklearn.model_selection import train_test_split
-from sklearn import linear_model
-
-def MSE(y_data,y_model):
- n = np.size(y_model)
- return np.sum((y_data-y_model)**2)/n
-
-def fit_beta(X, y):
- return np.linalg.pinv(X.T @ X) @ X.T @ y
-
-
-
-# A seed just to ensure that the random numbers are the same for every run.
-# Useful for eventual debugging.
-np.random.seed(3155)
-
-n = 100
-x = np.random.rand(n)
-y = 10.0+5*x
-
-
-Maxpolydegree = 2
-X = np.zeros((n,Maxpolydegree))
-
-for degree in range(Maxpolydegree):
- X[:,degree] = x**degree
-
-
-# We split the data in test and training data
-X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
-
-# Intercept is included in the design matrix
-# own code first
-OwnBeta = fit_beta(X_train, y_train)
-# Scikit-Learn
-skl = LinearRegression(fit_intercept=False).fit(X_train, y_train)
-ypredictOwn = X_test @ OwnBeta
-ypredictSKL = skl.predict(X_test)
-print(MSE(y_test,ypredictOwn)
-print(MSE(y_test,ypredictSKL)
-!ec
-
Printing the MSE, we see first that both methods give the same
MSE. However, changing the value of the intercept gives a larger or
@@ -597,7 +558,6 @@ function or simply scaling our results, leads to a fit which is
independent of the specific value of the intercept.
-
!split
===== Code Examples =====