diff --git a/doc/pub/DimRed/html/._DimRed-bs000.html b/doc/pub/DimRed/html/._DimRed-bs000.html index b87451a93..4d83066c4 100644 --- a/doc/pub/DimRed/html/._DimRed-bs000.html +++ b/doc/pub/DimRed/html/._DimRed-bs000.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -174,7 +179,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 19, 2019

    +

    Oct 20, 2019


    @@ -198,7 +203,7 @@ MathJax.Hub.Config({

  • 9
  • 10
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs001.html b/doc/pub/DimRed/html/._DimRed-bs001.html index 07788dd92..a9498ae16 100644 --- a/doc/pub/DimRed/html/._DimRed-bs001.html +++ b/doc/pub/DimRed/html/._DimRed-bs001.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -198,7 +203,7 @@ data.
  • 10
  • 11
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs002.html b/doc/pub/DimRed/html/._DimRed-bs002.html index 40e9fbfa3..13ecbc2ab 100644 --- a/doc/pub/DimRed/html/._DimRed-bs002.html +++ b/doc/pub/DimRed/html/._DimRed-bs002.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -197,7 +202,7 @@ ensures that all features are exactly between \( 0 \) and \( 1 \). The
  • 11
  • 12
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs003.html b/doc/pub/DimRed/html/._DimRed-bs003.html index 1ec740a66..2ec698d22 100644 --- a/doc/pub/DimRed/html/._DimRed-bs003.html +++ b/doc/pub/DimRed/html/._DimRed-bs003.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -200,7 +205,7 @@ techniques.
  • 12
  • 13
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs004.html b/doc/pub/DimRed/html/._DimRed-bs004.html index d5e0a2e1f..a06f5ea25 100644 --- a/doc/pub/DimRed/html/._DimRed-bs004.html +++ b/doc/pub/DimRed/html/._DimRed-bs004.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -275,7 +280,7 @@ svm.fit(X_train_scaled, y_train)
  • 13
  • 14
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs005.html b/doc/pub/DimRed/html/._DimRed-bs005.html index b9d0cd7d3..818201cdb 100644 --- a/doc/pub/DimRed/html/._DimRed-bs005.html +++ b/doc/pub/DimRed/html/._DimRed-bs005.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -225,7 +230,7 @@ svm.fit(X_train_scaled, y_train)
  • 14
  • 15
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs006.html b/doc/pub/DimRed/html/._DimRed-bs006.html index 60eec4c18..307eae100 100644 --- a/doc/pub/DimRed/html/._DimRed-bs006.html +++ b/doc/pub/DimRed/html/._DimRed-bs006.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -204,7 +209,7 @@ logreg.fit(X_train_scaled, y_train)
  • 15
  • 16
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs007.html b/doc/pub/DimRed/html/._DimRed-bs007.html index ce8c904ff..cd6de3817 100644 --- a/doc/pub/DimRed/html/._DimRed-bs007.html +++ b/doc/pub/DimRed/html/._DimRed-bs007.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -200,27 +205,41 @@ plt.show() #print eigvalues of correlation matrix EigValues, EigVectors = np.linalg.eig(correlation_matrix) print(EigValues) - -#split into train and test and then scale thereafter -X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0) -print(X_train.shape) -print(X_test.shape) - -logreg = LogisticRegression() -logreg.fit(X_train, y_train) -print("Test set accuracy from Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test))) - -from sklearn.preprocessing import MinMaxScaler, StandardScaler -scaler = StandardScaler() -scaler.fit(X_train) -X_train_scaled = scaler.transform(X_train) -X_test_scaled = scaler.transform(X_test) - -logreg.fit(X_train_scaled, y_train) -print("Test set accuracy scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))

    - +In the above example we note two things. In the first plot we display +the overlap of benign and malignant tumors as functions of the various +features in the Wisconsing breast cancer data set. We see that for +some of the features we can distinguish clearly the benign and +malignant cases while for other features we cannot. This can point to +us which features may be of greater interest when we wish to classify +a benign or not benign tumour. + +

    +In the second figure we have computed the so-called correlation +matrix, which in our case with thirty features becomes a \( 30\times 30 \) +matrix. + +

    +We constructed this matrix using pandas via the statements +

    + + +

    cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
    +
    +

    +and then +

    + + +

    correlation_matrix = cancerpd.corr().round(1)
    +
    +

    +Diagonalizing this matrix we can in turn say something about which +features are of relevance and which are not. But before we proceed we +need to define covariance and correlation matrices. This leads us to +the classical Principal Component Analysis (PCA) theorem with +applications.

    @@ -245,7 +264,7 @@ logreg.fit(X_train_scaled, y_train)

  • 16
  • 17
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs008.html b/doc/pub/DimRed/html/._DimRed-bs008.html index 61e98700f..6675e552a 100644 --- a/doc/pub/DimRed/html/._DimRed-bs008.html +++ b/doc/pub/DimRed/html/._DimRed-bs008.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -181,7 +186,7 @@ MathJax.Hub.Config({
  • 17
  • 18
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs009.html b/doc/pub/DimRed/html/._DimRed-bs009.html index 8481e8279..cac6aa00b 100644 --- a/doc/pub/DimRed/html/._DimRed-bs009.html +++ b/doc/pub/DimRed/html/._DimRed-bs009.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -157,6 +162,48 @@ MathJax.Hub.Config({

    Introducing the Covariance and Correlation functions

    +

    +Suppose we have defined two vectors +$\hat{x} and \hat{y} with \( n \) elements each. The covariance matrix $\boldsymbol{C}is defined as +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} cov[\boldsymbol{x},\boldsymbol{x}] & cov[\boldsymbol{x},\boldsymbol{y}] \\ + cov[\boldsymbol{y},\boldsymbol{x}] & cov[\boldsymbol{y},\boldsymbol{y}] \\ + \end{bmatrix}, +$$ + +where for example +$$ +cov[\boldsymbol{x},\boldsymbol{y}] =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +$$ + +With this definition and recalling that the variance is +$$ +var[\boldsymbol{x}]=\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})^2, +$$ + +we can rewrite the covariance matrix in this case as +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} var[\boldsymbol{x}] & cov[\boldsymbol{x},\boldsymbol{y}] \\ + cov[\boldsymbol{x},\boldsymbol{y}] & var[\boldsymbol{y}] \\ + \end{bmatrix}, +$$ + +

    +The covariance takes values between zero and infinity and may thus lead to problems with loss of numerical precision for particularly large values. It is common to scale the covariance matrix by introducing instead the correlation matrix defined via the so-called correlation function +$$ +corr[\boldsymbol{x},\boldsymbol{y}]=\frac{cov[\boldsymbol{x},\boldsymbol{y}]}{\sqrt{var[\boldsymbol{x}]\var\boldsymbol{y}]}}. +$$ + +The correlation function is then given by values \( corr[\boldsymbol{x},\boldsymbol{y}] \in [-1,1] \). This avoids eventual problems with too large values. We can then define the correlation matrix for the two vectors \( \boldsymbol{x} \) and \( \boldsymbol{y} \) as +$$ +\boldsymbol{K}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} 1 & corr[\boldsymbol{x},\boldsymbol{y}] \\ + corr[\boldsymbol{y},\boldsymbol{x}] & 1 \\ + \end{bmatrix}, +$$ + +

    +In the above example this is the function we constructed using pandas. +

    @@ -182,7 +229,7 @@ MathJax.Hub.Config({

  • 18
  • 19
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/DimRed-bs.html b/doc/pub/DimRed/html/DimRed-bs.html index b87451a93..4d83066c4 100644 --- a/doc/pub/DimRed/html/DimRed-bs.html +++ b/doc/pub/DimRed/html/DimRed-bs.html @@ -72,17 +72,21 @@ Automatically generated HTML file from DocOnce source 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -129,17 +133,18 @@ MathJax.Hub.Config({
  • Why should we think of reducing the dimensionality
  • Basic ideas of the Principal Component Analysis (PCA)
  • Introducing the Covariance and Correlation functions
  • -
  • Classical PCA Theorem
  • -
  • Prof of the PCA Theorem
  • -
  • Getting started with PCA
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Correlation Function and Design/Feature Matrix
  • +
  • Classical PCA Theorem
  • +
  • Prof of the PCA Theorem
  • +
  • Getting started with PCA
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -174,7 +179,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 19, 2019

    +

    Oct 20, 2019


    @@ -198,7 +203,7 @@ MathJax.Hub.Config({

  • 9
  • 10
  • ...
  • -
  • 21
  • +
  • 22
  • »
  • diff --git a/doc/pub/DimRed/html/DimRed-reveal.html b/doc/pub/DimRed/html/DimRed-reveal.html index 82886c900..20496dcee 100644 --- a/doc/pub/DimRed/html/DimRed-reveal.html +++ b/doc/pub/DimRed/html/DimRed-reveal.html @@ -148,7 +148,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

     
    -

    Oct 19, 2019

    +

    Oct 20, 2019


    @@ -469,27 +469,41 @@ plt.show() #print eigvalues of correlation matrix EigValues, EigVectors = np.linalg.eig(correlation_matrix) print(EigValues) - -#split into train and test and then scale thereafter -X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0) -print(X_train.shape) -print(X_test.shape) - -logreg = LogisticRegression() -logreg.fit(X_train, y_train) -print("Test set accuracy from Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test))) - -from sklearn.preprocessing import MinMaxScaler, StandardScaler -scaler = StandardScaler() -scaler.fit(X_train) -X_train_scaled = scaler.transform(X_train) -X_test_scaled = scaler.transform(X_test) - -logreg.fit(X_train_scaled, y_train) -print("Test set accuracy scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))

    - +In the above example we note two things. In the first plot we display +the overlap of benign and malignant tumors as functions of the various +features in the Wisconsing breast cancer data set. We see that for +some of the features we can distinguish clearly the benign and +malignant cases while for other features we cannot. This can point to +us which features may be of greater interest when we wish to classify +a benign or not benign tumour. + +

    +In the second figure we have computed the so-called correlation +matrix, which in our case with thirty features becomes a \( 30\times 30 \) +matrix. + +

    +We constructed this matrix using pandas via the statements +

    + + +

    cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
    +
    +

    +and then +

    + + +

    correlation_matrix = cancerpd.corr().round(1)
    +
    +

    +Diagonalizing this matrix we can in turn say something about which +features are of relevance and which are not. But before we proceed we +need to define covariance and correlation matrices. This leads us to +the classical Principal Component Analysis (PCA) theorem with +applications. @@ -500,21 +514,163 @@ logreg.fit(X_train_scaled, y_train)

    Introducing the Covariance and Correlation functions

    + +

    +Suppose we have defined two vectors +$\hat{x} and \hat{y} with \( n \) elements each. The covariance matrix $\boldsymbol{C}is defined as +

     
    +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} cov[\boldsymbol{x},\boldsymbol{x}] & cov[\boldsymbol{x},\boldsymbol{y}] \\ + cov[\boldsymbol{y},\boldsymbol{x}] & cov[\boldsymbol{y},\boldsymbol{y}] \\ + \end{bmatrix}, +$$ +

     
    + +where for example +

     
    +$$ +cov[\boldsymbol{x},\boldsymbol{y}] =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +$$ +

     
    + +With this definition and recalling that the variance is +

     
    +$$ +var[\boldsymbol{x}]=\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})^2, +$$ +

     
    + +we can rewrite the covariance matrix in this case as +

     
    +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} var[\boldsymbol{x}] & cov[\boldsymbol{x},\boldsymbol{y}] \\ + cov[\boldsymbol{x},\boldsymbol{y}] & var[\boldsymbol{y}] \\ + \end{bmatrix}, +$$ +

     
    + +

    +The covariance takes values between zero and infinity and may thus lead to problems with loss of numerical precision for particularly large values. It is common to scale the covariance matrix by introducing instead the correlation matrix defined via the so-called correlation function +

     
    +$$ +corr[\boldsymbol{x},\boldsymbol{y}]=\frac{cov[\boldsymbol{x},\boldsymbol{y}]}{\sqrt{var[\boldsymbol{x}]\var\boldsymbol{y}]}}. +$$ +

     
    + +The correlation function is then given by values \( corr[\boldsymbol{x},\boldsymbol{y}] \in [-1,1] \). This avoids eventual problems with too large values. We can then define the correlation matrix for the two vectors \( \boldsymbol{x} \) and \( \boldsymbol{y} \) as +

     
    +$$ +\boldsymbol{K}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} 1 & corr[\boldsymbol{x},\boldsymbol{y}] \\ + corr[\boldsymbol{y},\boldsymbol{x}] & 1 \\ + \end{bmatrix}, +$$ +

     
    + +

    +In the above example this is the function we constructed using pandas.

    -

    Classical PCA Theorem

    +

    Correlation Function and Design/Feature Matrix

    + +

    +In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix \( \boldsymbol{X} \) as +

     
    +$$ +\boldsymbol{X}=\begin{bmatrix} +x_{0,0} & x_{0,1} & x_{0,2}& \dots & \dots x_{0,p-1}\\ +x_{1,0} & x_{1,1} & x_{1,2}& \dots & \dots x_{1,p-1}\\ +x_{2,0} & x_{2,1} & x_{2,2}& \dots & \dots x_{2,p-1}\\ +\dots & \dots & \dots & \dots \dots & \dots \\ +x_{n-2,0} & x_{n-2,1} & x_{n-2,2}& \dots & \dots x_{n-2,p-1}\\ +x_{n-1,0} & x_{n-1,1} & x_{n-1,2}& \dots & \dots x_{n-1,p-1}\\ +\end{bmatrix}, +$$ +

     
    + +with \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \), with the predictors/features \( p \) refering to the column numbers and the +entries \( n \) being the row elements. +We can rewrite the design/feature matrix in terms of its column vectors as +

     
    +$$ +\boldsymbol{X}=\begin{bmatrix} \boldsymbol{x}_0 & \boldsymbol{x}_0 & \boldsymbol{x}_0 & \dots & \dots & \boldsymbol{x}_{p-1}\end{bmatrix}, +$$ +

     
    + +with a given vector +

     
    +$$ +\boldsymbol{x}_i^T = \begin{bmatrix}x_{0,i} & x_{1,i} & x_{2,i}& \dots & \dots x_{n-1,i}\end{bmatrix}. +$$ +

     
    + +

    +With these definitions, we can now rewrite our \( 2\times 2 \) correaltion/covariance matrix in terms of a moe general design/feature matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i =0,1,\dots,p-1 \) +

     
    +$$ +\boldsymbol{C}[\boldsymbol{x}] = \begin{bmatrix} var[\boldsymbol{x}_0] & cov[\boldsymbol{x}_0,\boldsymbol{x}_1] \\ + cov[\boldsymbol{x},\boldsymbol{y}] & var[\boldsymbol{x}_{p-1}] \\ + \end{bmatrix}, +$$ +

     
    + +

    +The Numpy function np.cov calculates the covariance elements using the factor \( 1/(n-1) \) instead of \( 1/n \) since it assumes we do not have the exact mean values. +The following simple function uses the np.vstack function which takes each vector of dimension \( 1\times n \) and produces a \( 2\times n \) matrix \( \hat{W} \) +

     
    +$$ +\hat{W} = \begin{bmatrix} x_0 & y_0 \\ + x_1 & y_1 \\ + x_2 & y_2\\ + \dots & \dots \\ + x_{n-2} & y_{n-2}\\ + x_{n-1} & y_{n-1} & + \end{bmatrix}, +$$ +

     
    + +

    +which in turn is converted into into the \( 3\times 3 \) covariance matrix +\( \hat{\Sigma} \) via the Numpy function np.cov(). We note that we can also calculate +the mean value of each set of samples \( \hat{x} \) etc using the Numpy +function np.mean(x). We can also extract the eigenvalues of the +covariance matrix through the np.linalg.eig() function. + +

    + + +

    # Importing various packages
    +import numpy as np
    +
    +n = 100
    +x = np.random.normal(size=n)
    +print(np.mean(x))
    +y = 4+3*x+np.random.normal(size=n)
    +print(np.mean(y))
    +z = x**3+np.random.normal(size=n)
    +print(np.mean(z))
    +W = np.vstack((x, y, z))
    +Sigma = np.cov(W)
    +print(Sigma)
    +Eigvals, Eigvecs = np.linalg.eig(Sigma)
    +print(Eigvals)
    +
    -

    Prof of the PCA Theorem

    +

    Classical PCA Theorem

    -

    Getting started with PCA

    +

    Prof of the PCA Theorem

    +
    + + +
    +

    Getting started with PCA

    @@ -530,7 +686,7 @@ X_pca = pca.transform(X_train_scaled)

    -

    Principal Component Analysis

    +

    Principal Component Analysis

    @@ -567,7 +723,7 @@ X2D = X_centered.dot(W2)

    -

    PCA and scikit-learn

    +

    PCA and scikit-learn

    Scikit-Learn’s PCA class implements PCA using SVD decomposition just like we did before. The @@ -598,7 +754,7 @@ More material to come here.

    -

    More on the PCA

    +

    More on the PCA

    Instead of arbitrarily choosing the number of dimensions to reduce down to, it is generally preferable to @@ -629,7 +785,7 @@ X_reduced = pca.fit_transform(X)

    -

    Incremental PCA

    +

    Incremental PCA

    One problem with the preceding implementation of PCA is that it requires the whole training set to fit in @@ -641,7 +797,7 @@ instances arrive).

    -

    Randomized PCA

    +

    Randomized PCA

    Scikit-Learn offers yet another option to perform PCA, called Randomized PCA. This is a stochastic @@ -655,7 +811,7 @@ previous algorithms when \( d \) is much smaller than \( n \).

    -

    Kernel PCA

    +

    Kernel PCA

    @@ -681,7 +837,7 @@ X_reduced = rbf_pca.fit_transform(X)

    -

    LLE

    +

    LLE

    Locally Linear Embedding (LLE) is another very powerful nonlinear dimensionality reduction @@ -693,7 +849,7 @@ these local relationships are best preserved (more details shortly).

    -

    Other techniques

    +

    Other techniques

    There are many other dimensionality reduction techniques, several of which are available in Scikit-Learn. @@ -707,6 +863,63 @@ Here are some of the most popular:

  • t-Distributed Stochastic Neighbor Embedding (t-SNE) reduces dimensionality while trying to keep similar instances close and dissimilar instances apart. It is mostly used for visualization, in particular to visualize clusters of instances in high-dimensional space (e.g., to visualize the MNIST images in 2D).
  • Linear Discriminant Analysis (LDA) is actually a classification algorithm, but during training it learns the most discriminative axes between the classes, and these axes can then be used to define a hyperplane onto which to project the data. The benefit is that the projection will keep classes as far apart as possible, so LDA is a good technique to reduce dimensionality before running another classification algorithm such as a Support Vector Machine (SVM) classifier discussed in the SVM lectures.
  • +

    + +Here are other examples where we use the DataFrame functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix +of dimensionality \( 10\times 5 \) and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations. +

    + + +

    import numpy as np
    +import pandas as pd
    +from IPython.display import display
    +np.random.seed(100)
    +# setting up a 10 x 5 matrix
    +rows = 10
    +cols = 5
    +a = np.random.randn(rows,cols)
    +df = pd.DataFrame(a)
    +display(df)
    +print(df.mean())
    +print(df.std())
    +display(df**2)
    +
    +

    +Thereafter we can select specific columns only and plot final results +

    + + +

    df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']
    +df.index = np.arange(10)
    +
    +display(df)
    +print(df['Second'].mean() )
    +print(df.info())
    +print(df.describe())
    +
    +from pylab import plt, mpl
    +plt.style.use('seaborn')
    +mpl.rcParams['font.family'] = 'serif'
    +
    +df.cumsum().plot(lw=2.0, figsize=(10,6))
    +plt.show()
    +
    +
    +df.plot.bar(figsize=(10,6), rot=15)
    +plt.show()
    +
    +

    +We can produce a \( 4\times 4 \) matrix +

    + + +

    b = np.arange(16).reshape((4,4))
    +print(b)
    +df1 = pd.DataFrame(b)
    +print(df1)
    +
    +

    +and many other operations.

    diff --git a/doc/pub/DimRed/html/DimRed-solarized.html b/doc/pub/DimRed/html/DimRed-solarized.html index f777e2e44..6cf01e88c 100644 --- a/doc/pub/DimRed/html/DimRed-solarized.html +++ b/doc/pub/DimRed/html/DimRed-solarized.html @@ -92,17 +92,21 @@ div { text-align: justify; text-justify: inter-word; } 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -144,7 +148,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 19, 2019

    +

    Oct 20, 2019












    @@ -462,27 +466,41 @@ plt.show() #print eigvalues of correlation matrix EigValues, EigVectors = np.linalg.eig(correlation_matrix) print(EigValues) - -#split into train and test and then scale thereafter -X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0) -print(X_train.shape) -print(X_test.shape) - -logreg = LogisticRegression() -logreg.fit(X_train, y_train) -print("Test set accuracy from Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test))) - -from sklearn.preprocessing import MinMaxScaler, StandardScaler -scaler = StandardScaler() -scaler.fit(X_train) -X_train_scaled = scaler.transform(X_train) -X_test_scaled = scaler.transform(X_test) - -logreg.fit(X_train_scaled, y_train) -print("Test set accuracy scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))

    - +In the above example we note two things. In the first plot we display +the overlap of benign and malignant tumors as functions of the various +features in the Wisconsing breast cancer data set. We see that for +some of the features we can distinguish clearly the benign and +malignant cases while for other features we cannot. This can point to +us which features may be of greater interest when we wish to classify +a benign or not benign tumour. + +

    +In the second figure we have computed the so-called correlation +matrix, which in our case with thirty features becomes a \( 30\times 30 \) +matrix. + +

    +We constructed this matrix using pandas via the statements +

    + + +

    cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
    +
    +

    +and then +

    + + +

    correlation_matrix = cancerpd.corr().round(1)
    +
    +

    +Diagonalizing this matrix we can in turn say something about which +features are of relevance and which are not. But before we proceed we +need to define covariance and correlation matrices. This leads us to +the classical Principal Component Analysis (PCA) theorem with +applications.











    @@ -495,19 +513,138 @@ logreg.fit(X_train_scaled, y_train)

    Introducing the Covariance and Correlation functions

    -









    +Suppose we have defined two vectors +$\hat{x} and \hat{y} with \( n \) elements each. The covariance matrix $\boldsymbol{C}is defined as +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} cov[\boldsymbol{x},\boldsymbol{x}] & cov[\boldsymbol{x},\boldsymbol{y}] \\ + cov[\boldsymbol{y},\boldsymbol{x}] & cov[\boldsymbol{y},\boldsymbol{y}] \\ + \end{bmatrix}, +$$ -

    Classical PCA Theorem

    +where for example +$$ +cov[\boldsymbol{x},\boldsymbol{y}] =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +$$ + +With this definition and recalling that the variance is +$$ +var[\boldsymbol{x}]=\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})^2, +$$ + +we can rewrite the covariance matrix in this case as +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} var[\boldsymbol{x}] & cov[\boldsymbol{x},\boldsymbol{y}] \\ + cov[\boldsymbol{x},\boldsymbol{y}] & var[\boldsymbol{y}] \\ + \end{bmatrix}, +$$ + +

    +The covariance takes values between zero and infinity and may thus lead to problems with loss of numerical precision for particularly large values. It is common to scale the covariance matrix by introducing instead the correlation matrix defined via the so-called correlation function +$$ +corr[\boldsymbol{x},\boldsymbol{y}]=\frac{cov[\boldsymbol{x},\boldsymbol{y}]}{\sqrt{var[\boldsymbol{x}]\var\boldsymbol{y}]}}. +$$ + +The correlation function is then given by values \( corr[\boldsymbol{x},\boldsymbol{y}] \in [-1,1] \). This avoids eventual problems with too large values. We can then define the correlation matrix for the two vectors \( \boldsymbol{x} \) and \( \boldsymbol{y} \) as +$$ +\boldsymbol{K}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} 1 & corr[\boldsymbol{x},\boldsymbol{y}] \\ + corr[\boldsymbol{y},\boldsymbol{x}] & 1 \\ + \end{bmatrix}, +$$ + +

    +In the above example this is the function we constructed using pandas.











    -

    Prof of the PCA Theorem

    +

    Correlation Function and Design/Feature Matrix

    + +

    +In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix \( \boldsymbol{X} \) as +$$ +\boldsymbol{X}=\begin{bmatrix} +x_{0,0} & x_{0,1} & x_{0,2}& \dots & \dots x_{0,p-1}\\ +x_{1,0} & x_{1,1} & x_{1,2}& \dots & \dots x_{1,p-1}\\ +x_{2,0} & x_{2,1} & x_{2,2}& \dots & \dots x_{2,p-1}\\ +\dots & \dots & \dots & \dots \dots & \dots \\ +x_{n-2,0} & x_{n-2,1} & x_{n-2,2}& \dots & \dots x_{n-2,p-1}\\ +x_{n-1,0} & x_{n-1,1} & x_{n-1,2}& \dots & \dots x_{n-1,p-1}\\ +\end{bmatrix}, +$$ + +with \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \), with the predictors/features \( p \) refering to the column numbers and the +entries \( n \) being the row elements. +We can rewrite the design/feature matrix in terms of its column vectors as +$$ +\boldsymbol{X}=\begin{bmatrix} \boldsymbol{x}_0 & \boldsymbol{x}_0 & \boldsymbol{x}_0 & \dots & \dots & \boldsymbol{x}_{p-1}\end{bmatrix}, +$$ + +with a given vector +$$ +\boldsymbol{x}_i^T = \begin{bmatrix}x_{0,i} & x_{1,i} & x_{2,i}& \dots & \dots x_{n-1,i}\end{bmatrix}. +$$ + +

    +With these definitions, we can now rewrite our \( 2\times 2 \) correaltion/covariance matrix in terms of a moe general design/feature matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i =0,1,\dots,p-1 \) +$$ +\boldsymbol{C}[\boldsymbol{x}] = \begin{bmatrix} var[\boldsymbol{x}_0] & cov[\boldsymbol{x}_0,\boldsymbol{x}_1] \\ + cov[\boldsymbol{x},\boldsymbol{y}] & var[\boldsymbol{x}_{p-1}] \\ + \end{bmatrix}, +$$ + +

    +The Numpy function np.cov calculates the covariance elements using the factor \( 1/(n-1) \) instead of \( 1/n \) since it assumes we do not have the exact mean values. +The following simple function uses the np.vstack function which takes each vector of dimension \( 1\times n \) and produces a \( 2\times n \) matrix \( \hat{W} \) +$$ +\hat{W} = \begin{bmatrix} x_0 & y_0 \\ + x_1 & y_1 \\ + x_2 & y_2\\ + \dots & \dots \\ + x_{n-2} & y_{n-2}\\ + x_{n-1} & y_{n-1} & + \end{bmatrix}, +$$ + +

    +which in turn is converted into into the \( 3\times 3 \) covariance matrix +\( \hat{\Sigma} \) via the Numpy function np.cov(). We note that we can also calculate +the mean value of each set of samples \( \hat{x} \) etc using the Numpy +function np.mean(x). We can also extract the eigenvalues of the +covariance matrix through the np.linalg.eig() function. + +

    + + +

    # Importing various packages
    +import numpy as np
    +
    +n = 100
    +x = np.random.normal(size=n)
    +print(np.mean(x))
    +y = 4+3*x+np.random.normal(size=n)
    +print(np.mean(y))
    +z = x**3+np.random.normal(size=n)
    +print(np.mean(z))
    +W = np.vstack((x, y, z))
    +Sigma = np.cov(W)
    +print(Sigma)
    +Eigvals, Eigvecs = np.linalg.eig(Sigma)
    +print(Eigvals)
    +
    +

    +









    + +

    Classical PCA Theorem











    -

    Getting started with PCA

    +

    Prof of the PCA Theorem

    + +

    +









    + +

    Getting started with PCA

    @@ -522,7 +659,7 @@ X_pca = pca.transform(X_train_scaled)











    -

    Principal Component Analysis

    +

    Principal Component Analysis

    @@ -558,7 +695,7 @@ X2D = X_centered.dot(W2)

    -

    PCA and scikit-learn

    +

    PCA and scikit-learn

    Scikit-Learn’s PCA class implements PCA using SVD decomposition just like we did before. The @@ -589,7 +726,7 @@ More material to come here.











    -

    More on the PCA

    +

    More on the PCA

    Instead of arbitrarily choosing the number of dimensions to reduce down to, it is generally preferable to @@ -619,7 +756,7 @@ X_reduced = pca.fit_transform(X)











    -

    Incremental PCA

    +

    Incremental PCA

    One problem with the preceding implementation of PCA is that it requires the whole training set to fit in @@ -631,7 +768,7 @@ instances arrive).











    -

    Randomized PCA

    +

    Randomized PCA

    Scikit-Learn offers yet another option to perform PCA, called Randomized PCA. This is a stochastic @@ -646,7 +783,7 @@ previous algorithms when \( d \) is much smaller than \( n \).











    -

    Kernel PCA

    +

    Kernel PCA

    @@ -675,7 +812,7 @@ X_reduced = rbf_pca.fit_transform(X)











    -

    LLE

    +

    LLE

    Locally Linear Embedding (LLE) is another very powerful nonlinear dimensionality reduction @@ -687,7 +824,7 @@ these local relationships are best preserved (more details shortly).











    -

    Other techniques

    +

    Other techniques

    There are many other dimensionality reduction techniques, several of which are available in Scikit-Learn. @@ -702,6 +839,61 @@ Here are some of the most popular:

  • Linear Discriminant Analysis (LDA) is actually a classification algorithm, but during training it learns the most discriminative axes between the classes, and these axes can then be used to define a hyperplane onto which to project the data. The benefit is that the projection will keep classes as far apart as possible, so LDA is a good technique to reduce dimensionality before running another classification algorithm such as a Support Vector Machine (SVM) classifier discussed in the SVM lectures.
  • +Here are other examples where we use the DataFrame functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix +of dimensionality \( 10\times 5 \) and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations. +

    + + +

    import numpy as np
    +import pandas as pd
    +from IPython.display import display
    +np.random.seed(100)
    +# setting up a 10 x 5 matrix
    +rows = 10
    +cols = 5
    +a = np.random.randn(rows,cols)
    +df = pd.DataFrame(a)
    +display(df)
    +print(df.mean())
    +print(df.std())
    +display(df**2)
    +
    +

    +Thereafter we can select specific columns only and plot final results +

    + + +

    df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']
    +df.index = np.arange(10)
    +
    +display(df)
    +print(df['Second'].mean() )
    +print(df.info())
    +print(df.describe())
    +
    +from pylab import plt, mpl
    +plt.style.use('seaborn')
    +mpl.rcParams['font.family'] = 'serif'
    +
    +df.cumsum().plot(lw=2.0, figsize=(10,6))
    +plt.show()
    +
    +
    +df.plot.bar(figsize=(10,6), rot=15)
    +plt.show()
    +
    +

    +We can produce a \( 4\times 4 \) matrix +

    + + +

    b = np.arange(16).reshape((4,4))
    +print(b)
    +df1 = pd.DataFrame(b)
    +print(df1)
    +
    +

    +and many other operations. diff --git a/doc/pub/DimRed/html/DimRed.html b/doc/pub/DimRed/html/DimRed.html index 16c9a74ea..ddbbd9055 100644 --- a/doc/pub/DimRed/html/DimRed.html +++ b/doc/pub/DimRed/html/DimRed.html @@ -97,17 +97,21 @@ div { text-align: justify; text-justify: inter-word; } 2, None, '___sec8'), - ('Classical PCA Theorem', 2, None, '___sec9'), - ('Prof of the PCA Theorem', 2, None, '___sec10'), - ('Getting started with PCA', 2, None, '___sec11'), - ('Principal Component Analysis', 2, None, '___sec12'), - ('PCA and scikit-learn', 2, None, '___sec13'), - ('More on the PCA', 2, None, '___sec14'), - ('Incremental PCA', 2, None, '___sec15'), - ('Randomized PCA', 2, None, '___sec16'), - ('Kernel PCA', 2, None, '___sec17'), - ('LLE', 2, None, '___sec18'), - ('Other techniques', 2, None, '___sec19')]} + ('Correlation Function and Design/Feature Matrix', + 2, + None, + '___sec9'), + ('Classical PCA Theorem', 2, None, '___sec10'), + ('Prof of the PCA Theorem', 2, None, '___sec11'), + ('Getting started with PCA', 2, None, '___sec12'), + ('Principal Component Analysis', 2, None, '___sec13'), + ('PCA and scikit-learn', 2, None, '___sec14'), + ('More on the PCA', 2, None, '___sec15'), + ('Incremental PCA', 2, None, '___sec16'), + ('Randomized PCA', 2, None, '___sec17'), + ('Kernel PCA', 2, None, '___sec18'), + ('LLE', 2, None, '___sec19'), + ('Other techniques', 2, None, '___sec20')]} end of tocinfo --> @@ -149,7 +153,7 @@ MathJax.Hub.Config({

    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 19, 2019

    +

    Oct 20, 2019












    @@ -467,27 +471,41 @@ plt.show() #print eigvalues of correlation matrix EigValues, EigVectors = np.linalg.eig(correlation_matrix) print(EigValues) - -#split into train and test and then scale thereafter -X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0) -print(X_train.shape) -print(X_test.shape) - -logreg = LogisticRegression() -logreg.fit(X_train, y_train) -print("Test set accuracy from Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test))) - -from sklearn.preprocessing import MinMaxScaler, StandardScaler -scaler = StandardScaler() -scaler.fit(X_train) -X_train_scaled = scaler.transform(X_train) -X_test_scaled = scaler.transform(X_test) - -logreg.fit(X_train_scaled, y_train) -print("Test set accuracy scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))

    - +In the above example we note two things. In the first plot we display +the overlap of benign and malignant tumors as functions of the various +features in the Wisconsing breast cancer data set. We see that for +some of the features we can distinguish clearly the benign and +malignant cases while for other features we cannot. This can point to +us which features may be of greater interest when we wish to classify +a benign or not benign tumour. + +

    +In the second figure we have computed the so-called correlation +matrix, which in our case with thirty features becomes a \( 30\times 30 \) +matrix. + +

    +We constructed this matrix using pandas via the statements +

    + + +

    cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
    +
    +

    +and then +

    + + +

    correlation_matrix = cancerpd.corr().round(1)
    +
    +

    +Diagonalizing this matrix we can in turn say something about which +features are of relevance and which are not. But before we proceed we +need to define covariance and correlation matrices. This leads us to +the classical Principal Component Analysis (PCA) theorem with +applications.











    @@ -500,19 +518,138 @@ logreg.fit(X_train_scaled, y_train)

    Introducing the Covariance and Correlation functions

    -









    +Suppose we have defined two vectors +$\hat{x} and \hat{y} with \( n \) elements each. The covariance matrix $\boldsymbol{C}is defined as +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} cov[\boldsymbol{x},\boldsymbol{x}] & cov[\boldsymbol{x},\boldsymbol{y}] \\ + cov[\boldsymbol{y},\boldsymbol{x}] & cov[\boldsymbol{y},\boldsymbol{y}] \\ + \end{bmatrix}, +$$ -

    Classical PCA Theorem

    +where for example +$$ +cov[\boldsymbol{x},\boldsymbol{y}] =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +$$ + +With this definition and recalling that the variance is +$$ +var[\boldsymbol{x}]=\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})^2, +$$ + +we can rewrite the covariance matrix in this case as +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} var[\boldsymbol{x}] & cov[\boldsymbol{x},\boldsymbol{y}] \\ + cov[\boldsymbol{x},\boldsymbol{y}] & var[\boldsymbol{y}] \\ + \end{bmatrix}, +$$ + +

    +The covariance takes values between zero and infinity and may thus lead to problems with loss of numerical precision for particularly large values. It is common to scale the covariance matrix by introducing instead the correlation matrix defined via the so-called correlation function +$$ +corr[\boldsymbol{x},\boldsymbol{y}]=\frac{cov[\boldsymbol{x},\boldsymbol{y}]}{\sqrt{var[\boldsymbol{x}]\var\boldsymbol{y}]}}. +$$ + +The correlation function is then given by values \( corr[\boldsymbol{x},\boldsymbol{y}] \in [-1,1] \). This avoids eventual problems with too large values. We can then define the correlation matrix for the two vectors \( \boldsymbol{x} \) and \( \boldsymbol{y} \) as +$$ +\boldsymbol{K}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} 1 & corr[\boldsymbol{x},\boldsymbol{y}] \\ + corr[\boldsymbol{y},\boldsymbol{x}] & 1 \\ + \end{bmatrix}, +$$ + +

    +In the above example this is the function we constructed using pandas.











    -

    Prof of the PCA Theorem

    +

    Correlation Function and Design/Feature Matrix

    + +

    +In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix \( \boldsymbol{X} \) as +$$ +\boldsymbol{X}=\begin{bmatrix} +x_{0,0} & x_{0,1} & x_{0,2}& \dots & \dots x_{0,p-1}\\ +x_{1,0} & x_{1,1} & x_{1,2}& \dots & \dots x_{1,p-1}\\ +x_{2,0} & x_{2,1} & x_{2,2}& \dots & \dots x_{2,p-1}\\ +\dots & \dots & \dots & \dots \dots & \dots \\ +x_{n-2,0} & x_{n-2,1} & x_{n-2,2}& \dots & \dots x_{n-2,p-1}\\ +x_{n-1,0} & x_{n-1,1} & x_{n-1,2}& \dots & \dots x_{n-1,p-1}\\ +\end{bmatrix}, +$$ + +with \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \), with the predictors/features \( p \) refering to the column numbers and the +entries \( n \) being the row elements. +We can rewrite the design/feature matrix in terms of its column vectors as +$$ +\boldsymbol{X}=\begin{bmatrix} \boldsymbol{x}_0 & \boldsymbol{x}_0 & \boldsymbol{x}_0 & \dots & \dots & \boldsymbol{x}_{p-1}\end{bmatrix}, +$$ + +with a given vector +$$ +\boldsymbol{x}_i^T = \begin{bmatrix}x_{0,i} & x_{1,i} & x_{2,i}& \dots & \dots x_{n-1,i}\end{bmatrix}. +$$ + +

    +With these definitions, we can now rewrite our \( 2\times 2 \) correaltion/covariance matrix in terms of a moe general design/feature matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i =0,1,\dots,p-1 \) +$$ +\boldsymbol{C}[\boldsymbol{x}] = \begin{bmatrix} var[\boldsymbol{x}_0] & cov[\boldsymbol{x}_0,\boldsymbol{x}_1] \\ + cov[\boldsymbol{x},\boldsymbol{y}] & var[\boldsymbol{x}_{p-1}] \\ + \end{bmatrix}, +$$ + +

    +The Numpy function np.cov calculates the covariance elements using the factor \( 1/(n-1) \) instead of \( 1/n \) since it assumes we do not have the exact mean values. +The following simple function uses the np.vstack function which takes each vector of dimension \( 1\times n \) and produces a \( 2\times n \) matrix \( \hat{W} \) +$$ +\hat{W} = \begin{bmatrix} x_0 & y_0 \\ + x_1 & y_1 \\ + x_2 & y_2\\ + \dots & \dots \\ + x_{n-2} & y_{n-2}\\ + x_{n-1} & y_{n-1} & + \end{bmatrix}, +$$ + +

    +which in turn is converted into into the \( 3\times 3 \) covariance matrix +\( \hat{\Sigma} \) via the Numpy function np.cov(). We note that we can also calculate +the mean value of each set of samples \( \hat{x} \) etc using the Numpy +function np.mean(x). We can also extract the eigenvalues of the +covariance matrix through the np.linalg.eig() function. + +

    + + +

    # Importing various packages
    +import numpy as np
    +
    +n = 100
    +x = np.random.normal(size=n)
    +print(np.mean(x))
    +y = 4+3*x+np.random.normal(size=n)
    +print(np.mean(y))
    +z = x**3+np.random.normal(size=n)
    +print(np.mean(z))
    +W = np.vstack((x, y, z))
    +Sigma = np.cov(W)
    +print(Sigma)
    +Eigvals, Eigvecs = np.linalg.eig(Sigma)
    +print(Eigvals)
    +
    +

    +









    + +

    Classical PCA Theorem











    -

    Getting started with PCA

    +

    Prof of the PCA Theorem

    + +

    +









    + +

    Getting started with PCA

    @@ -527,7 +664,7 @@ X_pca = pca.









    -

    Principal Component Analysis

    +

    Principal Component Analysis

    @@ -563,7 +700,7 @@ X2D = X_centered -

    PCA and scikit-learn

    +

    PCA and scikit-learn

    Scikit-Learn’s PCA class implements PCA using SVD decomposition just like we did before. The @@ -594,7 +731,7 @@ More material to come here.











    -

    More on the PCA

    +

    More on the PCA

    Instead of arbitrarily choosing the number of dimensions to reduce down to, it is generally preferable to @@ -624,7 +761,7 @@ X_reduced = pca











    -

    Incremental PCA

    +

    Incremental PCA

    One problem with the preceding implementation of PCA is that it requires the whole training set to fit in @@ -636,7 +773,7 @@ instances arrive).











    -

    Randomized PCA

    +

    Randomized PCA

    Scikit-Learn offers yet another option to perform PCA, called Randomized PCA. This is a stochastic @@ -651,7 +788,7 @@ previous algorithms when \( d \) is much smaller than \( n \).











    -

    Kernel PCA

    +

    Kernel PCA

    @@ -680,7 +817,7 @@ X_reduced = rbf_pcaLLE +

    LLE

    Locally Linear Embedding (LLE) is another very powerful nonlinear dimensionality reduction @@ -692,7 +829,7 @@ these local relationships are best preserved (more details shortly).











    -

    Other techniques

    +

    Other techniques

    There are many other dimensionality reduction techniques, several of which are available in Scikit-Learn. @@ -707,6 +844,61 @@ Here are some of the most popular:

  • Linear Discriminant Analysis (LDA) is actually a classification algorithm, but during training it learns the most discriminative axes between the classes, and these axes can then be used to define a hyperplane onto which to project the data. The benefit is that the projection will keep classes as far apart as possible, so LDA is a good technique to reduce dimensionality before running another classification algorithm such as a Support Vector Machine (SVM) classifier discussed in the SVM lectures.
  • +Here are other examples where we use the DataFrame functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix +of dimensionality \( 10\times 5 \) and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations. +

    + + +

    import numpy as np
    +import pandas as pd
    +from IPython.display import display
    +np.random.seed(100)
    +# setting up a 10 x 5 matrix
    +rows = 10
    +cols = 5
    +a = np.random.randn(rows,cols)
    +df = pd.DataFrame(a)
    +display(df)
    +print(df.mean())
    +print(df.std())
    +display(df**2)
    +
    +

    +Thereafter we can select specific columns only and plot final results +

    + + +

    df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']
    +df.index = np.arange(10)
    +
    +display(df)
    +print(df['Second'].mean() )
    +print(df.info())
    +print(df.describe())
    +
    +from pylab import plt, mpl
    +plt.style.use('seaborn')
    +mpl.rcParams['font.family'] = 'serif'
    +
    +df.cumsum().plot(lw=2.0, figsize=(10,6))
    +plt.show()
    +
    +
    +df.plot.bar(figsize=(10,6), rot=15)
    +plt.show()
    +
    +

    +We can produce a \( 4\times 4 \) matrix +

    + + +

    b = np.arange(16).reshape((4,4))
    +print(b)
    +df1 = pd.DataFrame(b)
    +print(df1)
    +
    +

    +and many other operations. diff --git a/doc/pub/DimRed/ipynb/DimRed.ipynb b/doc/pub/DimRed/ipynb/DimRed.ipynb index cc73fb1ff..e764465eb 100644 --- a/doc/pub/DimRed/ipynb/DimRed.ipynb +++ b/doc/pub/DimRed/ipynb/DimRed.ipynb @@ -10,7 +10,7 @@ " \n", "**Morten Hjorth-Jensen**, Department of Physics, University of Oslo and Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University\n", "\n", - "Date: **Oct 19, 2019**\n", + "Date: **Oct 20, 2019**\n", "\n", "Copyright 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license\n", "\n", @@ -337,32 +337,67 @@ "\n", "#print eigvalues of correlation matrix\n", "EigValues, EigVectors = np.linalg.eig(correlation_matrix)\n", - "print(EigValues)\n", - "\n", - "#split into train and test and then scale thereafter\n", - "X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)\n", - "print(X_train.shape)\n", - "print(X_test.shape)\n", - "\n", - "logreg = LogisticRegression()\n", - "logreg.fit(X_train, y_train)\n", - "print(\"Test set accuracy from Logistic Regression: {:.2f}\".format(logreg.score(X_test,y_test)))\n", - "\n", - "from sklearn.preprocessing import MinMaxScaler, StandardScaler\n", - "scaler = StandardScaler()\n", - "scaler.fit(X_train)\n", - "X_train_scaled = scaler.transform(X_train)\n", - "X_test_scaled = scaler.transform(X_test)\n", - "\n", - "logreg.fit(X_train_scaled, y_train)\n", - "print(\"Test set accuracy scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))" + "print(EigValues)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ - "\n", + "In the above example we note two things. In the first plot we display\n", + "the overlap of benign and malignant tumors as functions of the various\n", + "features in the Wisconsing breast cancer data set. We see that for\n", + "some of the features we can distinguish clearly the benign and\n", + "malignant cases while for other features we cannot. This can point to\n", + "us which features may be of greater interest when we wish to classify\n", + "a benign or not benign tumour.\n", + "\n", + "In the second figure we have computed the so-called correlation\n", + "matrix, which in our case with thirty features becomes a $30\\times 30$\n", + "matrix.\n", + "\n", + "We constructed this matrix using **pandas** via the statements" + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and then" + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "correlation_matrix = cancerpd.corr().round(1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Diagonalizing this matrix we can in turn say something about which\n", + "features are of relevance and which are not. But before we proceed we\n", + "need to define covariance and correlation matrices. This leads us to\n", + "the classical Principal Component Analysis (PCA) theorem with\n", + "applications.\n", + "\n", "\n", "\n", "## Basic ideas of the Principal Component Analysis (PCA)\n", @@ -370,6 +405,247 @@ "\n", "## Introducing the Covariance and Correlation functions\n", "\n", + "Suppose we have defined two vectors\n", + "$\\hat{x} and \\hat{y} with $n$ elements each. The covariance matrix $\\boldsymbol{C}is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x},\\boldsymbol{y}] = \\begin{bmatrix} cov[\\boldsymbol{x},\\boldsymbol{x}] & cov[\\boldsymbol{x},\\boldsymbol{y}] \\\\\n", + " cov[\\boldsymbol{y},\\boldsymbol{x}] & cov[\\boldsymbol{y},\\boldsymbol{y}] \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where for example" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "cov[\\boldsymbol{x},\\boldsymbol{y}] =\\frac{1}{n} \\sum_{i=0}^{n-1}(x_i- \\overline{x})(y_i- \\overline{y}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With this definition and recalling that the variance is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "var[\\boldsymbol{x}]=\\frac{1}{n} \\sum_{i=0}^{n-1}(x_i- \\overline{x})^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we can rewrite the covariance matrix in this case as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x},\\boldsymbol{y}] = \\begin{bmatrix} var[\\boldsymbol{x}] & cov[\\boldsymbol{x},\\boldsymbol{y}] \\\\\n", + " cov[\\boldsymbol{x},\\boldsymbol{y}] & var[\\boldsymbol{y}] \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The covariance takes values between zero and infinity and may thus lead to problems with loss of numerical precision for particularly large values. It is common to scale the covariance matrix by introducing instead the correlation matrix defined via the so-called correlation function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "corr[\\boldsymbol{x},\\boldsymbol{y}]=\\frac{cov[\\boldsymbol{x},\\boldsymbol{y}]}{\\sqrt{var[\\boldsymbol{x}]\\var\\boldsymbol{y}]}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The correlation function is then given by values $corr[\\boldsymbol{x},\\boldsymbol{y}] \\in [-1,1]$. This avoids eventual problems with too large values. We can then define the correlation matrix for the two vectors $\\boldsymbol{x}$ and $\\boldsymbol{y}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{K}[\\boldsymbol{x},\\boldsymbol{y}] = \\begin{bmatrix} 1 & corr[\\boldsymbol{x},\\boldsymbol{y}] \\\\\n", + " corr[\\boldsymbol{y},\\boldsymbol{x}] & 1 \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the above example this is the function we constructed using **pandas**.\n", + "\n", + "## Correlation Function and Design/Feature Matrix\n", + "\n", + "In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix $\\boldsymbol{X}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\\begin{bmatrix}\n", + "x_{0,0} & x_{0,1} & x_{0,2}& \\dots & \\dots x_{0,p-1}\\\\\n", + "x_{1,0} & x_{1,1} & x_{1,2}& \\dots & \\dots x_{1,p-1}\\\\\n", + "x_{2,0} & x_{2,1} & x_{2,2}& \\dots & \\dots x_{2,p-1}\\\\\n", + "\\dots & \\dots & \\dots & \\dots \\dots & \\dots \\\\\n", + "x_{n-2,0} & x_{n-2,1} & x_{n-2,2}& \\dots & \\dots x_{n-2,p-1}\\\\\n", + "x_{n-1,0} & x_{n-1,1} & x_{n-1,2}& \\dots & \\dots x_{n-1,p-1}\\\\\n", + "\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$, with the predictors/features $p$ refering to the column numbers and the\n", + "entries $n$ being the row elements.\n", + "We can rewrite the design/feature matrix in terms of its column vectors as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\\begin{bmatrix} \\boldsymbol{x}_0 & \\boldsymbol{x}_0 & \\boldsymbol{x}_0 & \\dots & \\dots & \\boldsymbol{x}_{p-1}\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with a given vector" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_i^T = \\begin{bmatrix}x_{0,i} & x_{1,i} & x_{2,i}& \\dots & \\dots x_{n-1,i}\\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With these definitions, we can now rewrite our $2\\times 2$ correaltion/covariance matrix in terms of a moe general design/feature matrix $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$. This leads to a $p\\times p$ covariance matrix for the vectors $\\boldsymbol{x}_i$ with $i =0,1,\\dots,p-1$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x}] = \\begin{bmatrix} var[\\boldsymbol{x}_0] & cov[\\boldsymbol{x}_0,\\boldsymbol{x}_1] \\\\\n", + " cov[\\boldsymbol{x},\\boldsymbol{y}] & var[\\boldsymbol{x}_{p-1}] \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The Numpy function **np.cov** calculates the covariance elements using the factor $1/(n-1)$ instead of $1/n$ since it assumes we do not have the exact mean values. \n", + "The following simple function uses the **np.vstack** function which takes each vector of dimension $1\\times n$ and produces a $2\\times n$ matrix $\\hat{W}$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\hat{W} = \\begin{bmatrix} x_0 & y_0 \\\\\n", + " x_1 & y_1 \\\\\n", + " x_2 & y_2\\\\\n", + " \\dots & \\dots \\\\\n", + " x_{n-2} & y_{n-2}\\\\\n", + " x_{n-1} & y_{n-1} & \n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which in turn is converted into into the $3\\times 3$ covariance matrix\n", + "$\\hat{\\Sigma}$ via the Numpy function **np.cov()**. We note that we can also calculate\n", + "the mean value of each set of samples $\\hat{x}$ etc using the Numpy\n", + "function **np.mean(x)**. We can also extract the eigenvalues of the\n", + "covariance matrix through the **np.linalg.eig()** function." + ] + }, + { + "cell_type": "code", + "execution_count": 7, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "import numpy as np\n", + "\n", + "n = 100\n", + "x = np.random.normal(size=n)\n", + "print(np.mean(x))\n", + "y = 4+3*x+np.random.normal(size=n)\n", + "print(np.mean(y))\n", + "z = x**3+np.random.normal(size=n)\n", + "print(np.mean(z))\n", + "W = np.vstack((x, y, z))\n", + "Sigma = np.cov(W)\n", + "print(Sigma)\n", + "Eigvals, Eigvecs = np.linalg.eig(Sigma)\n", + "print(Eigvals)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ "## Classical PCA Theorem\n", "\n", "\n", @@ -385,7 +661,7 @@ }, { "cell_type": "code", - "execution_count": 5, + "execution_count": 8, "metadata": { "collapsed": false }, @@ -413,7 +689,7 @@ }, { "cell_type": "code", - "execution_count": 6, + "execution_count": 9, "metadata": { "collapsed": false }, @@ -440,7 +716,7 @@ }, { "cell_type": "code", - "execution_count": 7, + "execution_count": 10, "metadata": { "collapsed": false }, @@ -464,7 +740,7 @@ }, { "cell_type": "code", - "execution_count": 8, + "execution_count": 11, "metadata": { "collapsed": false }, @@ -486,7 +762,7 @@ }, { "cell_type": "code", - "execution_count": 9, + "execution_count": 12, "metadata": { "collapsed": false }, @@ -516,7 +792,7 @@ }, { "cell_type": "code", - "execution_count": 10, + "execution_count": 13, "metadata": { "collapsed": false }, @@ -539,7 +815,7 @@ }, { "cell_type": "code", - "execution_count": 11, + "execution_count": 14, "metadata": { "collapsed": false }, @@ -586,7 +862,7 @@ }, { "cell_type": "code", - "execution_count": 12, + "execution_count": 15, "metadata": { "collapsed": false }, @@ -623,7 +899,96 @@ "\n", "* **t-Distributed Stochastic Neighbor Embedding** (t-SNE) reduces dimensionality while trying to keep similar instances close and dissimilar instances apart. It is mostly used for visualization, in particular to visualize clusters of instances in high-dimensional space (e.g., to visualize the MNIST images in 2D).\n", "\n", - "* Linear Discriminant Analysis (LDA) is actually a classification algorithm, but during training it learns the most discriminative axes between the classes, and these axes can then be used to define a hyperplane onto which to project the data. The benefit is that the projection will keep classes as far apart as possible, so LDA is a good technique to reduce dimensionality before running another classification algorithm such as a Support Vector Machine (SVM) classifier discussed in the SVM lectures." + "* Linear Discriminant Analysis (LDA) is actually a classification algorithm, but during training it learns the most discriminative axes between the classes, and these axes can then be used to define a hyperplane onto which to project the data. The benefit is that the projection will keep classes as far apart as possible, so LDA is a good technique to reduce dimensionality before running another classification algorithm such as a Support Vector Machine (SVM) classifier discussed in the SVM lectures.\n", + "\n", + "Here are other examples where we use the **DataFrame** functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix \n", + "of dimensionality $10\\times 5$ and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations." + ] + }, + { + "cell_type": "code", + "execution_count": 16, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import pandas as pd\n", + "from IPython.display import display\n", + "np.random.seed(100)\n", + "# setting up a 10 x 5 matrix\n", + "rows = 10\n", + "cols = 5\n", + "a = np.random.randn(rows,cols)\n", + "df = pd.DataFrame(a)\n", + "display(df)\n", + "print(df.mean())\n", + "print(df.std())\n", + "display(df**2)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Thereafter we can select specific columns only and plot final results" + ] + }, + { + "cell_type": "code", + "execution_count": 17, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']\n", + "df.index = np.arange(10)\n", + "\n", + "display(df)\n", + "print(df['Second'].mean() )\n", + "print(df.info())\n", + "print(df.describe())\n", + "\n", + "from pylab import plt, mpl\n", + "plt.style.use('seaborn')\n", + "mpl.rcParams['font.family'] = 'serif'\n", + "\n", + "df.cumsum().plot(lw=2.0, figsize=(10,6))\n", + "plt.show()\n", + "\n", + "\n", + "df.plot.bar(figsize=(10,6), rot=15)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can produce a $4\\times 4$ matrix" + ] + }, + { + "cell_type": "code", + "execution_count": 18, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "b = np.arange(16).reshape((4,4))\n", + "print(b)\n", + "df1 = pd.DataFrame(b)\n", + "print(df1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and many other operations." ] } ], diff --git a/doc/pub/DimRed/ipynb/ipynb-DimRed-src.tar.gz b/doc/pub/DimRed/ipynb/ipynb-DimRed-src.tar.gz index cc8d3012b..559fe806e 100644 Binary files a/doc/pub/DimRed/ipynb/ipynb-DimRed-src.tar.gz and b/doc/pub/DimRed/ipynb/ipynb-DimRed-src.tar.gz differ diff --git a/doc/pub/DimRed/pdf/DimRed-minted.pdf b/doc/pub/DimRed/pdf/DimRed-minted.pdf index 85a5cc870..463adcf4d 100644 Binary files a/doc/pub/DimRed/pdf/DimRed-minted.pdf and b/doc/pub/DimRed/pdf/DimRed-minted.pdf differ diff --git a/doc/src/DimRed/DimRed.do.txt b/doc/src/DimRed/DimRed.do.txt index b29c5051f..a66916083 100644 --- a/doc/src/DimRed/DimRed.do.txt +++ b/doc/src/DimRed/DimRed.do.txt @@ -339,31 +339,91 @@ Suppose we have defined two vectors $\hat{x} and \hat{y} with $n$ elements each. The covariance matrix $\bm{C}is defined as !bt \[ -\bm{C}[\bm{x},\bm{y}] = \begin{bmatrix} cov[xx] & cov[xy] \\ - cov[yx] & cov[yy] \\ +\bm{C}[\bm{x},\bm{y}] = \begin{bmatrix} cov[\bm{x},\bm{x}] & cov[\bm{x},\bm{y}] \\ + cov[\bm{y},\bm{x}] & cov[\bm{y},\bm{y}] \\ \end{bmatrix}, \] !et where for example !bt \[ -cov[xy] =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +cov[\bm{x},\bm{y}] =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). \] !et With this definition and recalling that the variance is +!bt \[ -var[\bm{x}]=\sigma_{xx} =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})^2, +var[\bm{x}]=\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})^2, \] !et we can rewrite the covariance matrix in this case as !bt \[ -\bm{C}[\bm{x},\bm{y}] = \begin{bmatrix} var[\bm{x}] & \sigma_{xy} \\ - \sigma_{yx} & var[\bm{y}] \\ +\bm{C}[\bm{x},\bm{y}] = \begin{bmatrix} var[\bm{x}] & cov[\bm{x},\bm{y}] \\ + cov[\bm{x},\bm{y}] & var[\bm{y}] \\ \end{bmatrix}, \] !et +The covariance takes values between zero and infinity and may thus lead to problems with loss of numerical precision for particularly large values. It is common to scale the covariance matrix by introducing instead the correlation matrix defined via the so-called correlation function +!bt +\[ +corr[\bm{x},\bm{y}]=\frac{cov[\bm{x},\bm{y}]}{\sqrt{var[\bm{x}]\var\bm{y}]}}. +\] +!et +The correlation function is then given by values $corr[\bm{x},\bm{y}] \in [-1,1]$. This avoids eventual problems with too large values. We can then define the correlation matrix for the two vectors $\bm{x}$ and $\bm{y}$ as +!bt +\[ +\bm{K}[\bm{x},\bm{y}] = \begin{bmatrix} 1 & corr[\bm{x},\bm{y}] \\ + corr[\bm{y},\bm{x}] & 1 \\ + \end{bmatrix}, +\] +!et + +In the above example this is the function we constructed using _pandas_. + +!split +===== Correlation Function and Design/Feature Matrix ===== + +In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix $\bm{X}$ as +!bt +\[ +\bm{X}=\begin{bmatrix} +x_{0,0} & x_{0,1} & x_{0,2}& \dots & \dots x_{0,p-1}\\ +x_{1,0} & x_{1,1} & x_{1,2}& \dots & \dots x_{1,p-1}\\ +x_{2,0} & x_{2,1} & x_{2,2}& \dots & \dots x_{2,p-1}\\ +\dots & \dots & \dots & \dots \dots & \dots \\ +x_{n-2,0} & x_{n-2,1} & x_{n-2,2}& \dots & \dots x_{n-2,p-1}\\ +x_{n-1,0} & x_{n-1,1} & x_{n-1,2}& \dots & \dots x_{n-1,p-1}\\ +\end{bmatrix}, +\] +!et +with $\bm{X}\in {\mathbb{R}}^{n\times p}$, with the predictors/features $p$ refering to the column numbers and the +entries $n$ being the row elements. +We can rewrite the design/feature matrix in terms of its column vectors as +!bt +\[ +\bm{X}=\begin{bmatrix} \bm{x}_0 & \bm{x}_0 & \bm{x}_0 & \dots & \dots & \bm{x}_{p-1}\end{bmatrix}, +\] +!et +with a given vector +!bt +\[ +\bm{x}_i^T = \begin{bmatrix}x_{0,i} & x_{1,i} & x_{2,i}& \dots & \dots x_{n-1,i}\end{bmatrix}. +\] +!et + +With these definitions, we can now rewrite our $2\times 2$ correaltion/covariance matrix in terms of a moe general design/feature matrix $\bm{X}\in {\mathbb{R}}^{n\times p}$. This leads to a $p\times p$ covariance matrix for the vectors $\bm{x}_i$ with $i =0,1,\dots,p-1$ +!bt +\[ +\bm{C}[\bm{x}] = \begin{bmatrix} var[\bm{x}_0] & cov[\bm{x}_0,\bm{x}_1] \\ + cov[\bm{x},\bm{y}] & var[\bm{x}_{p-1}] \\ + \end{bmatrix}, +\] +!et + + + The Numpy function _np.cov_ calculates the covariance elements using the factor $1/(n-1)$ instead of $1/n$ since it assumes we do not have the exact mean values. The following simple function uses the _np.vstack_ function which takes each vector of dimension $1\times n$ and produces a $2\times n$ matrix $\hat{W}$