diff --git a/doc/pub/DimRed/html/._DimRed-bs000.html b/doc/pub/DimRed/html/._DimRed-bs000.html index c0e2e1fdb..fcef4f476 100644 --- a/doc/pub/DimRed/html/._DimRed-bs000.html +++ b/doc/pub/DimRed/html/._DimRed-bs000.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -218,7 +224,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Dec 28, 2019

    +

    Dec 29, 2019


    @@ -242,7 +248,7 @@ MathJax.Hub.Config({

  • 9
  • 10
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs001.html b/doc/pub/DimRed/html/._DimRed-bs001.html index 604451fe3..95607135f 100644 --- a/doc/pub/DimRed/html/._DimRed-bs001.html +++ b/doc/pub/DimRed/html/._DimRed-bs001.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -250,7 +256,7 @@ visualization.
  • 10
  • 11
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs002.html b/doc/pub/DimRed/html/._DimRed-bs002.html index f281284d9..020984f1f 100644 --- a/doc/pub/DimRed/html/._DimRed-bs002.html +++ b/doc/pub/DimRed/html/._DimRed-bs002.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -241,7 +247,7 @@ ensures that all features are exactly between \( 0 \) and \( 1 \). The
  • 11
  • 12
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs003.html b/doc/pub/DimRed/html/._DimRed-bs003.html index d2bfcf82f..974ed102f 100644 --- a/doc/pub/DimRed/html/._DimRed-bs003.html +++ b/doc/pub/DimRed/html/._DimRed-bs003.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -244,7 +250,7 @@ techniques.
  • 12
  • 13
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs004.html b/doc/pub/DimRed/html/._DimRed-bs004.html index b9946be9a..3c420051d 100644 --- a/doc/pub/DimRed/html/._DimRed-bs004.html +++ b/doc/pub/DimRed/html/._DimRed-bs004.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -319,7 +325,7 @@ svm.fit(X_train_scaled, y_train)
  • 13
  • 14
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs005.html b/doc/pub/DimRed/html/._DimRed-bs005.html index c1f31f395..c2ac3ee56 100644 --- a/doc/pub/DimRed/html/._DimRed-bs005.html +++ b/doc/pub/DimRed/html/._DimRed-bs005.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -269,7 +275,7 @@ svm.fit(X_train_scaled, y_train)
  • 14
  • 15
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs006.html b/doc/pub/DimRed/html/._DimRed-bs006.html index 2f361bebf..05eccf8b5 100644 --- a/doc/pub/DimRed/html/._DimRed-bs006.html +++ b/doc/pub/DimRed/html/._DimRed-bs006.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -248,7 +254,7 @@ logreg.fit(X_train_scaled, y_train)
  • 15
  • 16
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs007.html b/doc/pub/DimRed/html/._DimRed-bs007.html index 1c3b2e6e7..7efbdf565 100644 --- a/doc/pub/DimRed/html/._DimRed-bs007.html +++ b/doc/pub/DimRed/html/._DimRed-bs007.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -303,7 +309,7 @@ applications.
  • 16
  • 17
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs008.html b/doc/pub/DimRed/html/._DimRed-bs008.html index ec4603204..3949fe0c4 100644 --- a/doc/pub/DimRed/html/._DimRed-bs008.html +++ b/doc/pub/DimRed/html/._DimRed-bs008.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -201,6 +207,15 @@ MathJax.Hub.Config({

    Basic ideas of the Principal Component Analysis (PCA)

    +

    +The principal component analysis deals with the problem of fitting a +low-dimensional affine subspace \( S \) of dimension \( d \) much smaller than +the totaldimension \( D \) of the problem at hand (our data +set). Mathematically it can be formulated as a statistical problem or +a geometric problem. In our discussion of the theorem for the +classical PCA, we will stay with a statistical approach. This is also +what set the scene historically which for the PCA. +

    We have a data set defined by a design/feature matrix \( \boldsymbol{X} \) (see below for its definition) @@ -233,7 +248,7 @@ We have a data set defined by a design/feature matrix \( \boldsymbol{X} \) (see

  • 17
  • 18
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs009.html b/doc/pub/DimRed/html/._DimRed-bs009.html index 05ce56b2b..5494b0ae6 100644 --- a/doc/pub/DimRed/html/._DimRed-bs009.html +++ b/doc/pub/DimRed/html/._DimRed-bs009.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -203,7 +209,7 @@ MathJax.Hub.Config({

    Before we discuss the PCA theorem, we need to remind ourselves about -the definition of the covariance and the correlation function. +the definition of the covariance and the correlation function. These are quantities

    Suppose we have defined two vectors @@ -282,7 +288,7 @@ In the above example this is the function we constructed using pandas.

  • 18
  • 19
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs010.html b/doc/pub/DimRed/html/._DimRed-bs010.html index 9fcb9607f..0b33892e5 100644 --- a/doc/pub/DimRed/html/._DimRed-bs010.html +++ b/doc/pub/DimRed/html/._DimRed-bs010.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -202,7 +208,9 @@ MathJax.Hub.Config({

    Correlation Function and Design/Feature Matrix

    -In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix \( \boldsymbol{X} \) as +In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression +we defined the design/feature matrix \( \boldsymbol{X} \) as + $$ \boldsymbol{X}=\begin{bmatrix} x_{0,0} & x_{0,1} & x_{0,2}& \dots & \dots x_{0,p-1}\\ @@ -227,7 +235,11 @@ $$ $$

    -With these definitions, we can now rewrite our \( 2\times 2 \) correaltion/covariance matrix in terms of a moe general design/feature matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i =0,1,\dots,p-1 \) +With these definitions, we can now rewrite our \( 2\times 2 \) +correaltion/covariance matrix in terms of a moe general design/feature +matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) +covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i=0,1,\dots,p-1 \) + $$ \boldsymbol{C}[\boldsymbol{x}] = \begin{bmatrix} \mathrm{var}[\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_1] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_2] & \dots & \dots & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_{p-1}]\\ @@ -277,7 +289,7 @@ $$

  • 19
  • 20
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs011.html b/doc/pub/DimRed/html/._DimRed-bs011.html index 77d5296ca..7eefc658d 100644 --- a/doc/pub/DimRed/html/._DimRed-bs011.html +++ b/doc/pub/DimRed/html/._DimRed-bs011.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -265,7 +271,7 @@ C = np.c
  • 20
  • 21
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs012.html b/doc/pub/DimRed/html/._DimRed-bs012.html index 7a8763b59..d2d517c11 100644 --- a/doc/pub/DimRed/html/._DimRed-bs012.html +++ b/doc/pub/DimRed/html/._DimRed-bs012.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -267,7 +273,7 @@ The above procedure with numpy can be made more compact if we use pand
  • 21
  • 22
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs013.html b/doc/pub/DimRed/html/._DimRed-bs013.html index 2ab2ddff5..9a02ba499 100644 --- a/doc/pub/DimRed/html/._DimRed-bs013.html +++ b/doc/pub/DimRed/html/._DimRed-bs013.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -249,7 +255,7 @@ We expand this model to the Franke function discussed above.
  • 22
  • 23
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs014.html b/doc/pub/DimRed/html/._DimRed-bs014.html index d7c6c6753..7fc40d5f4 100644 --- a/doc/pub/DimRed/html/._DimRed-bs014.html +++ b/doc/pub/DimRed/html/._DimRed-bs014.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -256,8 +262,8 @@ columns since all matrix elements in the design matrix were set to one

    This means that the variance for these elements will be zero and will cause problems when we set up the correlation matrix. We can simply -drop these elements as follows and then construct the correlation -matrix. +drop these elements and construct a correlation +matrix without these elements.

    @@ -285,7 +291,7 @@ matrix.

  • 23
  • 24
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs015.html b/doc/pub/DimRed/html/._DimRed-bs015.html index 86ccf8390..c98730ba2 100644 --- a/doc/pub/DimRed/html/._DimRed-bs015.html +++ b/doc/pub/DimRed/html/._DimRed-bs015.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -265,7 +271,7 @@ It is easy to generalize this to a matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\t
  • 24
  • 25
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs016.html b/doc/pub/DimRed/html/._DimRed-bs016.html index 6f742428c..983a50bab 100644 --- a/doc/pub/DimRed/html/._DimRed-bs016.html +++ b/doc/pub/DimRed/html/._DimRed-bs016.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -273,7 +279,7 @@ features/predictors.
  • 25
  • 26
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs017.html b/doc/pub/DimRed/html/._DimRed-bs017.html index 3d8d8fdb9..6392f4885 100644 --- a/doc/pub/DimRed/html/._DimRed-bs017.html +++ b/doc/pub/DimRed/html/._DimRed-bs017.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -253,7 +259,7 @@ $$
  • 26
  • 27
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs018.html b/doc/pub/DimRed/html/._DimRed-bs018.html index 10609edc8..9796a01be 100644 --- a/doc/pub/DimRed/html/._DimRed-bs018.html +++ b/doc/pub/DimRed/html/._DimRed-bs018.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -229,7 +235,7 @@ X = np.r Make thereafter a small Python code which plots the data. Note that the function multivariate returns also the covariance discussed above and that it is defined by dividing by \( n-1 \) instead of \( n \).

    -Now we are going to implement the PCA algorithm. We will break it down into sub-steps and across multiple cells. +Now we are going to implement the PCA algorithm. We will break it down into various substeps.

    Compute the sample mean and center the data

    @@ -319,7 +325,7 @@ Finally, try out your own PCA function with other data sets.
  • 27
  • 28
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs019.html b/doc/pub/DimRed/html/._DimRed-bs019.html index cce5ff3ab..0281b99d7 100644 --- a/doc/pub/DimRed/html/._DimRed-bs019.html +++ b/doc/pub/DimRed/html/._DimRed-bs019.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -202,8 +208,11 @@ MathJax.Hub.Config({

    Classical PCA Theorem

    -We assume now that we have a design matrix \( \boldsymbol{X} \) which has been centered as discussed above. For the sake of simplicity we skip the overline symbol. The matrix is defined in terms of the various column vectors \( [\boldsymbol{x}_0,\boldsymbol{x}_1,\dots, \boldsymbol{x}_{p-1}] \) -each with dimension \( \boldsymbol{x}\in {\mathbb{R}}^{n} \). +We assume now that we have a design matrix \( \boldsymbol{X} \) which has been +centered as discussed above. For the sake of simplicity we skip the +overline symbol. The matrix is defined in terms of the various column +vectors \( [\boldsymbol{x}_0,\boldsymbol{x}_1,\dots, \boldsymbol{x}_{p-1}] \) each with dimension +\( \boldsymbol{x}\in {\mathbb{R}}^{n} \).

    We assume also that we have an orthogonal transformation \( \boldsymbol{W}\in {\mathbb{R}}^{p\times p} \). We define the reconstruction error (which is similar to the mean squared error we have seen before) as @@ -215,7 +224,13 @@ with \( \overline{\boldsymbol{x}}_i = \boldsymbol{W}\boldsymbol{z}_i \), where \ \( \boldsymbol{Z}\in{\mathbb{R}}^{p\times n} \). When doing PCA we want to reduce this dimensionality.

    -The PCA theorem states that minimizing the above reconstruction error corresponds to setting \( \boldsymbol{W}=\boldsymbol{S} \), the orthogonal matrix which diagonalizes the empirical covariance(correlation) matrix. The optimal low-dimensional encoding of the data is then given by a set of vectors \( \boldsymbol{z}_i \) with at most \( l \) vectors, with \( l < < p \), defined by the orthogonal projection of the data onto the columns spanned by the eigenvectors of the covariance(correlations matrix). +The PCA theorem states that minimizing the above reconstruction error +corresponds to setting \( \boldsymbol{W}=\boldsymbol{S} \), the orthogonal matrix which +diagonalizes the empirical covariance(correlation) matrix. The optimal +low-dimensional encoding of the data is then given by a set of vectors +\( \boldsymbol{z}_i \) with at most \( l \) vectors, with \( l < < p \), defined by the +orthogonal projection of the data onto the columns spanned by the +eigenvectors of the covariance(correlations matrix).

    @@ -243,7 +258,7 @@ The PCA theorem states that minimizing the above reconstruction error correspond

  • 28
  • 29
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs020.html b/doc/pub/DimRed/html/._DimRed-bs020.html index 536339c85..08aa77aa6 100644 --- a/doc/pub/DimRed/html/._DimRed-bs020.html +++ b/doc/pub/DimRed/html/._DimRed-bs020.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -245,7 +251,7 @@ where the vectors on the rhs are known.
  • 29
  • 30
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs021.html b/doc/pub/DimRed/html/._DimRed-bs021.html index 2ecf2204b..3a436e092 100644 --- a/doc/pub/DimRed/html/._DimRed-bs021.html +++ b/doc/pub/DimRed/html/._DimRed-bs021.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -232,7 +238,10 @@ $$ $$

    -We are almost there, we have obtained a relation between minimizing the reconstruction error and the variance and the covariance matrix. Minimizing the error is equivalent to maximizing the variance of the projected data. +We are almost there, we have obtained a relation between minimizing +the reconstruction error and the variance and the covariance +matrix. Minimizing the error is equivalent to maximizing the variance +of the projected data.

    @@ -260,7 +269,7 @@ We are almost there, we have obtained a relation between minimizing the reconstr

  • 30
  • 31
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs022.html b/doc/pub/DimRed/html/._DimRed-bs022.html index 285b05f8e..29dc48b88 100644 --- a/doc/pub/DimRed/html/._DimRed-bs022.html +++ b/doc/pub/DimRed/html/._DimRed-bs022.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -247,6 +253,9 @@ discussion in chapter 12.2 of Murphy's text has also a nice link with the Singular Value Decomposition theorem. For categorical data, see chapter 12.4 and discussion therein. +

    +Additional part of the proof for the other eigenvectors will be added by mid January 2020. +

    @@ -272,6 +281,8 @@ chapter 12.4 and discussion therein.

  • 30
  • 31
  • 32
  • +
  • ...
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs023.html b/doc/pub/DimRed/html/._DimRed-bs023.html index 67b68fa28..9f8585645 100644 --- a/doc/pub/DimRed/html/._DimRed-bs023.html +++ b/doc/pub/DimRed/html/._DimRed-bs023.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -199,59 +205,11 @@ MathJax.Hub.Config({ -

    Principal Component Analysis

    -
    -
    -

    -Principal Component Analysis (PCA) is by far the most popular dimensionality reduction algorithm. -First it identifies the hyperplane that lies closest to the data, and then it projects the data onto it. +

    Geometric Interpretation and link with Singular Value Decomposition

    -The following Python code uses NumPy’s svd() function to obtain all the principal components of the -training set, then extracts the first two principal components. First we center the data using either pandas or our own code -

    +This material will be added by mid January 2020. - -

    import numpy as np
    -import pandas as pd
    -from IPython.display import display
    -np.random.seed(100)
    -# setting up a 10 x 5 vanilla matrix 
    -rows = 10
    -cols = 5
    -X = np.random.randn(rows,cols)
    -df = pd.DataFrame(X)
    -# Pandas does the centering for us
    -df = df -df.mean()
    -display(df)
    -
    -# we center it ourselves
    -X_centered = X - X.mean(axis=0)
    -# Then check the difference between pandas and our own set up
    -print(X_centered-df)
    -#Now we do an SVD
    -U, s, V = np.linalg.svd(X_centered)
    -c1 = V.T[:, 0]
    -c2 = V.T[:, 1]
    -W2 = V.T[:, :2]
    -X2D = X_centered.dot(W2)
    -print(X2D)
    -
    -

    -PCA assumes that the dataset is centered around the origin. Scikit-Learn’s PCA classes take care of centering -the data for you. However, if you implement PCA yourself (as in the preceding example), or if you use other libraries, don’t -forget to center the data first. - -

    -Once you have identified all the principal components, you can reduce the dimensionality of the dataset -down to \( d \) dimensions by projecting it onto the hyperplane defined by the first \( d \) principal components. -Selecting this hyperplane ensures that the projection will preserve as much variance as possible. -

    - - -

    W2 = V.T[:, :2]
    -X2D = X_centered.dot(W2)
    -

    @@ -276,6 +234,7 @@ X2D = X_centered30

  • 31
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs024.html b/doc/pub/DimRed/html/._DimRed-bs024.html index 2e151571d..2be659898 100644 --- a/doc/pub/DimRed/html/._DimRed-bs024.html +++ b/doc/pub/DimRed/html/._DimRed-bs024.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -197,37 +203,61 @@ MathJax.Hub.Config({

     

     

     

    - + -

    PCA and scikit-learn

    +

    Principal Component Analysis

    +
    +
    +

    +Principal Component Analysis (PCA) is by far the most popular dimensionality reduction algorithm. +First it identifies the hyperplane that lies closest to the data, and then it projects the data onto it.

    -Scikit-Learn’s PCA class implements PCA using SVD decomposition just like we did before. The -following code applies PCA to reduce the dimensionality of the dataset down to two dimensions (note -that it automatically takes care of centering the data): +The following Python code uses NumPy’s svd() function to obtain all the principal components of the +training set, then extracts the first two principal components. First we center the data using either pandas or our own code

    -

    #thereafter we do a PCA with Scikit-learn
    -from sklearn.decomposition import PCA
    -pca = PCA(n_components = 2)
    -X2D = pca.fit_transform(X)
    +
    import numpy as np
    +import pandas as pd
    +from IPython.display import display
    +np.random.seed(100)
    +# setting up a 10 x 5 vanilla matrix 
    +rows = 10
    +cols = 5
    +X = np.random.randn(rows,cols)
    +df = pd.DataFrame(X)
    +# Pandas does the centering for us
    +df = df -df.mean()
    +display(df)
    +
    +# we center it ourselves
    +X_centered = X - X.mean(axis=0)
    +# Then check the difference between pandas and our own set up
    +print(X_centered-df)
    +#Now we do an SVD
    +U, s, V = np.linalg.svd(X_centered)
    +c1 = V.T[:, 0]
    +c2 = V.T[:, 1]
    +W2 = V.T[:, :2]
    +X2D = X_centered.dot(W2)
     print(X2D)
     

    -After fitting the PCA transformer to the dataset, you can access the principal components using the -components variable (note that it contains the PCs as horizontal vectors, so, for example, the first -principal component is equal to +PCA assumes that the dataset is centered around the origin. Scikit-Learn’s PCA classes take care of centering +the data for you. However, if you implement PCA yourself (as in the preceding example), or if you use other libraries, don’t +forget to center the data first. + +

    +Once you have identified all the principal components, you can reduce the dimensionality of the dataset +down to \( d \) dimensions by projecting it onto the hyperplane defined by the first \( d \) principal components. +Selecting this hyperplane ensures that the projection will preserve as much variance as possible.

    -

    pca.components_.T[:, 0].
    +
    W2 = V.T[:, :2]
    +X2D = X_centered.dot(W2)
     
    -

    -Another very useful piece of information is the explained variance ratio of each principal component, -available via the \( explained\_variance\_ratio \) variable. It indicates the proportion of the dataset’s -variance that lies along the axis of each principal component. -

    @@ -251,6 +281,7 @@ variance that lies along the axis of each principal component.

  • 30
  • 31
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs025.html b/doc/pub/DimRed/html/._DimRed-bs025.html index a1f2ee2be..f9daa99eb 100644 --- a/doc/pub/DimRed/html/._DimRed-bs025.html +++ b/doc/pub/DimRed/html/._DimRed-bs025.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -197,45 +203,36 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Back to the Cancer Data

    -We can now repeat the above but applied to real data, in this case our breast cancer data. -Here we compute performance scores on the training data using logistic regression. +

    PCA and scikit-learn

    + +

    +Scikit-Learn’s PCA class implements PCA using SVD decomposition just like we did before. The +following code applies PCA to reduce the dimensionality of the dataset down to two dimensions (note +that it automatically takes care of centering the data):

    -

    import matplotlib.pyplot as plt
    -import numpy as np
    -from sklearn.model_selection import  train_test_split 
    -from sklearn.datasets import load_breast_cancer
    -from sklearn.linear_model import LogisticRegression
    -cancer = load_breast_cancer()
    -
    -X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
    -
    -logreg = LogisticRegression()
    -logreg.fit(X_train, y_train)
    -print("Train set accuracy from Logistic Regression: {:.2f}".format(logreg.score(X_train,y_train)))
    -# We scale the data
    -from sklearn.preprocessing import StandardScaler
    -scaler = StandardScaler()
    -scaler.fit(X_train)
    -X_train_scaled = scaler.transform(X_train)
    -X_test_scaled = scaler.transform(X_test)
    -# Then perform again a log reg fit
    -logreg.fit(X_train_scaled, y_train)
    -print("Train set accuracy scaled data: {:.2f}".format(logreg.score(X_train_scaled,y_train)))
    -#thereafter we do a PCA with Scikit-learn
    +
    #thereafter we do a PCA with Scikit-learn
     from sklearn.decomposition import PCA
     pca = PCA(n_components = 2)
    -X2D_train = pca.fit_transform(X_train_scaled)
    -# and finally compute the log reg fit and the score on the training data	
    -logreg.fit(X2D_train,y_train)
    -print("Train set accuracy scaled and PCA data: {:.2f}".format(logreg.score(X2D_train,y_train)))
    +X2D = pca.fit_transform(X)
    +print(X2D)
     

    -We see that our training data after the PCA decomposition has a performance similar to the non-scaled data. +After fitting the PCA transformer to the dataset, you can access the principal components using the +components variable (note that it contains the PCs as horizontal vectors, so, for example, the first +principal component is equal to +

    + + +

    pca.components_.T[:, 0].
    +
    +

    +Another very useful piece of information is the explained variance ratio of each principal component, +available via the \( explained\_variance\_ratio \) variable. It indicates the proportion of the dataset’s +variance that lies along the axis of each principal component.

    @@ -259,6 +256,7 @@ We see that our training data after the PCA decomposition has a performance simi

  • 30
  • 31
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs026.html b/doc/pub/DimRed/html/._DimRed-bs026.html index 99b82253e..e59ea0bfd 100644 --- a/doc/pub/DimRed/html/._DimRed-bs026.html +++ b/doc/pub/DimRed/html/._DimRed-bs026.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -199,33 +205,44 @@ MathJax.Hub.Config({ -

    More on the PCA

    - -

    -Instead of arbitrarily choosing the number of dimensions to reduce down to, it is generally preferable to -choose the number of dimensions that add up to a sufficiently large portion of the variance (e.g., 95%). -Unless, of course, you are reducing dimensionality for data visualization — in that case you will -generally want to reduce the dimensionality down to 2 or 3. -The following code computes PCA without reducing dimensionality, then computes the minimum number -of dimensions required to preserve 95% of the training set’s variance: +

    Back to the Cancer Data

    +We can now repeat the above but applied to real data, in this case our breast cancer data. +Here we compute performance scores on the training data using logistic regression.

    -

    pca = PCA()
    -pca.fit(X)
    -cumsum = np.cumsum(pca.explained_variance_ratio_)
    -d = np.argmax(cumsum >= 0.95) + 1
    -
    -

    -You could then set \( n\_components=d \) and run PCA again. However, there is a much better option: instead -of specifying the number of principal components you want to preserve, you can set \( n\_components \) to be -a float between 0.0 and 1.0, indicating the ratio of variance you wish to preserve: -

    +

    import matplotlib.pyplot as plt
    +import numpy as np
    +from sklearn.model_selection import  train_test_split 
    +from sklearn.datasets import load_breast_cancer
    +from sklearn.linear_model import LogisticRegression
    +cancer = load_breast_cancer()
     
    -
    -
    pca = PCA(n_components=0.95)
    -X_reduced = pca.fit_transform(X)
    +X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
    +
    +logreg = LogisticRegression()
    +logreg.fit(X_train, y_train)
    +print("Train set accuracy from Logistic Regression: {:.2f}".format(logreg.score(X_train,y_train)))
    +# We scale the data
    +from sklearn.preprocessing import StandardScaler
    +scaler = StandardScaler()
    +scaler.fit(X_train)
    +X_train_scaled = scaler.transform(X_train)
    +X_test_scaled = scaler.transform(X_test)
    +# Then perform again a log reg fit
    +logreg.fit(X_train_scaled, y_train)
    +print("Train set accuracy scaled data: {:.2f}".format(logreg.score(X_train_scaled,y_train)))
    +#thereafter we do a PCA with Scikit-learn
    +from sklearn.decomposition import PCA
    +pca = PCA(n_components = 2)
    +X2D_train = pca.fit_transform(X_train_scaled)
    +# and finally compute the log reg fit and the score on the training data	
    +logreg.fit(X2D_train,y_train)
    +print("Train set accuracy scaled and PCA data: {:.2f}".format(logreg.score(X2D_train,y_train)))
     
    +

    +We see that our training data after the PCA decomposition has a performance similar to the non-scaled data. +

    @@ -247,6 +264,7 @@ X_reduced = pca

  • 30
  • 31
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs027.html b/doc/pub/DimRed/html/._DimRed-bs027.html index 95a52b810..56acdc987 100644 --- a/doc/pub/DimRed/html/._DimRed-bs027.html +++ b/doc/pub/DimRed/html/._DimRed-bs027.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -199,15 +205,33 @@ MathJax.Hub.Config({ -

    Incremental PCA

    +

    More on the PCA

    -One problem with the preceding implementation of PCA is that it requires the whole training set to fit in -memory in order for the SVD algorithm to run. Fortunately, Incremental PCA (IPCA) algorithms have -been developed: you can split the training set into mini-batches and feed an IPCA algorithm one minibatch -at a time. This is useful for large training sets, and also to apply PCA online (i.e., on the fly, as new -instances arrive). +Instead of arbitrarily choosing the number of dimensions to reduce down to, it is generally preferable to +choose the number of dimensions that add up to a sufficiently large portion of the variance (e.g., 95%). +Unless, of course, you are reducing dimensionality for data visualization — in that case you will +generally want to reduce the dimensionality down to 2 or 3. +The following code computes PCA without reducing dimensionality, then computes the minimum number +of dimensions required to preserve 95% of the training set’s variance: +

    + +

    pca = PCA()
    +pca.fit(X)
    +cumsum = np.cumsum(pca.explained_variance_ratio_)
    +d = np.argmax(cumsum >= 0.95) + 1
    +
    +

    +You could then set \( n\_components=d \) and run PCA again. However, there is a much better option: instead +of specifying the number of principal components you want to preserve, you can set \( n\_components \) to be +a float between 0.0 and 1.0, indicating the ratio of variance you wish to preserve: +

    + + +

    pca = PCA(n_components=0.95)
    +X_reduced = pca.fit_transform(X)
    +

    @@ -228,6 +252,7 @@ instances arrive).

  • 30
  • 31
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs028.html b/doc/pub/DimRed/html/._DimRed-bs028.html index 0c9c7f308..0e6a50276 100644 --- a/doc/pub/DimRed/html/._DimRed-bs028.html +++ b/doc/pub/DimRed/html/._DimRed-bs028.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -199,18 +205,14 @@ MathJax.Hub.Config({ -

    Randomized PCA

    +

    Incremental PCA

    -Scikit-Learn offers yet another option to perform PCA, called Randomized PCA. This is a stochastic -algorithm that quickly finds an approximation of the first d principal components. Its computational -complexity is \( O(m \times d^2)+O(d^3) \), instead of \( O(m \times n^2) + O(n^3) \), so it is dramatically faster than the -previous algorithms when \( d \) is much smaller than \( n \). - -

    -

    -
    - +One problem with the preceding implementation of PCA is that it requires the whole training set to fit in +memory in order for the SVD algorithm to run. Fortunately, Incremental PCA (IPCA) algorithms have +been developed: you can split the training set into mini-batches and feed an IPCA algorithm one minibatch +at a time. This is useful for large training sets, and also to apply PCA online (i.e., on the fly, as new +instances arrive).

    @@ -231,6 +233,7 @@ previous algorithms when \( d \) is much smaller than \( n \).

  • 30
  • 31
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs029.html b/doc/pub/DimRed/html/._DimRed-bs029.html index e605c052c..1c29effce 100644 --- a/doc/pub/DimRed/html/._DimRed-bs029.html +++ b/doc/pub/DimRed/html/._DimRed-bs029.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -199,28 +205,14 @@ MathJax.Hub.Config({ -

    Kernel PCA

    -
    -
    -

    +

    Randomized PCA

    -The kernel trick is a mathematical technique that implicitly maps instances into a -very high-dimensional space (called the feature space), enabling nonlinear classification and regression -with Support Vector Machines. Recall that a linear decision boundary in the high-dimensional feature -space corresponds to a complex nonlinear decision boundary in the original space. -It turns out that the same trick can be applied to PCA, making it possible to perform complex nonlinear -projections for dimensionality reduction. This is called Kernel PCA (kPCA). It is often good at -preserving clusters of instances after projection, or sometimes even unrolling datasets that lie close to a -twisted manifold. -For example, the following code uses Scikit-Learn’s KernelPCA class to perform kPCA with an -

    +Scikit-Learn offers yet another option to perform PCA, called Randomized PCA. This is a stochastic +algorithm that quickly finds an approximation of the first d principal components. Its computational +complexity is \( O(m \times d^2)+O(d^3) \), instead of \( O(m \times n^2) + O(n^3) \), so it is dramatically faster than the +previous algorithms when \( d \) is much smaller than \( n \). - -

    from sklearn.decomposition import KernelPCA
    -rbf_pca = KernelPCA(n_components = 2, kernel="rbf", gamma=0.04)
    -X_reduced = rbf_pca.fit_transform(X)
    -

    @@ -244,6 +236,7 @@ X_reduced = rbf_pca30
  • 31
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/._DimRed-bs030.html b/doc/pub/DimRed/html/._DimRed-bs030.html index 283c55794..f9bca9bdc 100644 --- a/doc/pub/DimRed/html/._DimRed-bs030.html +++ b/doc/pub/DimRed/html/._DimRed-bs030.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -199,14 +205,32 @@ MathJax.Hub.Config({ -

    LLE

    +

    Kernel PCA

    +
    +
    +

    -Locally Linear Embedding (LLE) is another very powerful nonlinear dimensionality reduction -(NLDR) technique. It is a Manifold Learning technique that does not rely on projections like the previous -algorithms. In a nutshell, LLE works by first measuring how each training instance linearly relates to its -closest neighbors (c.n.), and then looking for a low-dimensional representation of the training set where -these local relationships are best preserved (more details shortly). +The kernel trick is a mathematical technique that implicitly maps instances into a +very high-dimensional space (called the feature space), enabling nonlinear classification and regression +with Support Vector Machines. Recall that a linear decision boundary in the high-dimensional feature +space corresponds to a complex nonlinear decision boundary in the original space. +It turns out that the same trick can be applied to PCA, making it possible to perform complex nonlinear +projections for dimensionality reduction. This is called Kernel PCA (kPCA). It is often good at +preserving clusters of instances after projection, or sometimes even unrolling datasets that lie close to a +twisted manifold. +For example, the following code uses Scikit-Learn’s KernelPCA class to perform kPCA with an +

    + + +

    from sklearn.decomposition import KernelPCA
    +rbf_pca = KernelPCA(n_components = 2, kernel="rbf", gamma=0.04)
    +X_reduced = rbf_pca.fit_transform(X)
    +
    +

    +

    +
    +

    @@ -225,6 +249,7 @@ these local relationships are best preserved (more details shortly).

  • 30
  • 31
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/DimRed-bs.html b/doc/pub/DimRed/html/DimRed-bs.html index c0e2e1fdb..fcef4f476 100644 --- a/doc/pub/DimRed/html/DimRed-bs.html +++ b/doc/pub/DimRed/html/DimRed-bs.html @@ -104,15 +104,20 @@ Automatically generated HTML file from DocOnce source ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -175,15 +180,16 @@ MathJax.Hub.Config({
  • Proof of the PCA Theorem
  • PCA Proof continued
  • The final step
  • -
  • Principal Component Analysis
  • -
  • PCA and scikit-learn
  • -
  • Back to the Cancer Data
  • -
  • More on the PCA
  • -
  • Incremental PCA
  • -
  • Randomized PCA
  • -
  • Kernel PCA
  • -
  • LLE
  • -
  • Other techniques
  • +
  • Geometric Interpretation and link with Singular Value Decomposition
  • +
  • Principal Component Analysis
  • +
  • PCA and scikit-learn
  • +
  • Back to the Cancer Data
  • +
  • More on the PCA
  • +
  • Incremental PCA
  • +
  • Randomized PCA
  • +
  • Kernel PCA
  • +
  • LLE
  • +
  • Other techniques
  • @@ -218,7 +224,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Dec 28, 2019

    +

    Dec 29, 2019


    @@ -242,7 +248,7 @@ MathJax.Hub.Config({

  • 9
  • 10
  • ...
  • -
  • 32
  • +
  • 33
  • »
  • diff --git a/doc/pub/DimRed/html/DimRed-reveal.html b/doc/pub/DimRed/html/DimRed-reveal.html index 589de6f51..9c2731ba5 100644 --- a/doc/pub/DimRed/html/DimRed-reveal.html +++ b/doc/pub/DimRed/html/DimRed-reveal.html @@ -148,7 +148,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

     
    -

    Dec 28, 2019

    +

    Dec 29, 2019


    @@ -518,6 +518,15 @@ applications.

    Basic ideas of the Principal Component Analysis (PCA)

    +

    +The principal component analysis deals with the problem of fitting a +low-dimensional affine subspace \( S \) of dimension \( d \) much smaller than +the totaldimension \( D \) of the problem at hand (our data +set). Mathematically it can be formulated as a statistical problem or +a geometric problem. In our discussion of the theorem for the +classical PCA, we will stay with a statistical approach. This is also +what set the scene historically which for the PCA. +

    We have a data set defined by a design/feature matrix \( \boldsymbol{X} \) (see below for its definition) @@ -534,7 +543,7 @@ We have a data set defined by a design/feature matrix \( \boldsymbol{X} \) (see

    Before we discuss the PCA theorem, we need to remind ourselves about -the definition of the covariance and the correlation function. +the definition of the covariance and the correlation function. These are quantities

    Suppose we have defined two vectors @@ -606,7 +615,9 @@ In the above example this is the function we constructed using pandas.

    Correlation Function and Design/Feature Matrix

    -In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix \( \boldsymbol{X} \) as +In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression +we defined the design/feature matrix \( \boldsymbol{X} \) as +

     
    $$ \boldsymbol{X}=\begin{bmatrix} @@ -637,7 +648,11 @@ $$

     

    -With these definitions, we can now rewrite our \( 2\times 2 \) correaltion/covariance matrix in terms of a moe general design/feature matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i =0,1,\dots,p-1 \) +With these definitions, we can now rewrite our \( 2\times 2 \) +correaltion/covariance matrix in terms of a moe general design/feature +matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) +covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i=0,1,\dots,p-1 \) +

     
    $$ \boldsymbol{C}[\boldsymbol{x}] = \begin{bmatrix} @@ -843,8 +858,8 @@ columns since all matrix elements in the design matrix were set to one

    This means that the variance for these elements will be zero and will cause problems when we set up the correlation matrix. We can simply -drop these elements as follows and then construct the correlation -matrix. +drop these elements and construct a correlation +matrix without these elements.

    @@ -1026,7 +1041,7 @@ X = np.random.multivariate_normal(mean, cov, n) Make thereafter a small Python code which plots the data. Note that the function multivariate returns also the covariance discussed above and that it is defined by dividing by \( n-1 \) instead of \( n \).

    -Now we are going to implement the PCA algorithm. We will break it down into sub-steps and across multiple cells. +Now we are going to implement the PCA algorithm. We will break it down into various substeps.

    Compute the sample mean and center the data

    @@ -1103,8 +1118,11 @@ Finally, try out your own PCA function with other data sets.

    Classical PCA Theorem

    -We assume now that we have a design matrix \( \boldsymbol{X} \) which has been centered as discussed above. For the sake of simplicity we skip the overline symbol. The matrix is defined in terms of the various column vectors \( [\boldsymbol{x}_0,\boldsymbol{x}_1,\dots, \boldsymbol{x}_{p-1}] \) -each with dimension \( \boldsymbol{x}\in {\mathbb{R}}^{n} \). +We assume now that we have a design matrix \( \boldsymbol{X} \) which has been +centered as discussed above. For the sake of simplicity we skip the +overline symbol. The matrix is defined in terms of the various column +vectors \( [\boldsymbol{x}_0,\boldsymbol{x}_1,\dots, \boldsymbol{x}_{p-1}] \) each with dimension +\( \boldsymbol{x}\in {\mathbb{R}}^{n} \).

    We assume also that we have an orthogonal transformation \( \boldsymbol{W}\in {\mathbb{R}}^{p\times p} \). We define the reconstruction error (which is similar to the mean squared error we have seen before) as @@ -1118,7 +1136,13 @@ with \( \overline{\boldsymbol{x}}_i = \boldsymbol{W}\boldsymbol{z}_i \), where \ \( \boldsymbol{Z}\in{\mathbb{R}}^{p\times n} \). When doing PCA we want to reduce this dimensionality.

    -The PCA theorem states that minimizing the above reconstruction error corresponds to setting \( \boldsymbol{W}=\boldsymbol{S} \), the orthogonal matrix which diagonalizes the empirical covariance(correlation) matrix. The optimal low-dimensional encoding of the data is then given by a set of vectors \( \boldsymbol{z}_i \) with at most \( l \) vectors, with \( l < < p \), defined by the orthogonal projection of the data onto the columns spanned by the eigenvectors of the covariance(correlations matrix). +The PCA theorem states that minimizing the above reconstruction error +corresponds to setting \( \boldsymbol{W}=\boldsymbol{S} \), the orthogonal matrix which +diagonalizes the empirical covariance(correlation) matrix. The optimal +low-dimensional encoding of the data is then given by a set of vectors +\( \boldsymbol{z}_i \) with at most \( l \) vectors, with \( l < < p \), defined by the +orthogonal projection of the data onto the columns spanned by the +eigenvectors of the covariance(correlations matrix). @@ -1195,7 +1219,10 @@ $$

     

    -We are almost there, we have obtained a relation between minimizing the reconstruction error and the variance and the covariance matrix. Minimizing the error is equivalent to maximizing the variance of the projected data. +We are almost there, we have obtained a relation between minimizing +the reconstruction error and the variance and the covariance +matrix. Minimizing the error is equivalent to maximizing the variance +of the projected data. @@ -1255,11 +1282,22 @@ our basis of eigenvectors is orthogonal, see Principal Component Analysis +

    Geometric Interpretation and link with Singular Value Decomposition

    + +

    +This material will be added by mid January 2020. + + + +

    +

    Principal Component Analysis

    @@ -1316,7 +1354,7 @@ X2D = X_centered.dot(W2)

    -

    PCA and scikit-learn

    +

    PCA and scikit-learn

    Scikit-Learn’s PCA class implements PCA using SVD decomposition just like we did before. The @@ -1348,7 +1386,7 @@ variance that lies along the axis of each principal component.

    -

    Back to the Cancer Data

    +

    Back to the Cancer Data

    We can now repeat the above but applied to real data, in this case our breast cancer data. Here we compute performance scores on the training data using logistic regression.

    @@ -1389,7 +1427,7 @@ We see that our training data after the PCA decomposition has a performance simi

    -

    More on the PCA

    +

    More on the PCA

    Instead of arbitrarily choosing the number of dimensions to reduce down to, it is generally preferable to @@ -1420,7 +1458,7 @@ X_reduced = pca.fit_transform(X)

    -

    Incremental PCA

    +

    Incremental PCA

    One problem with the preceding implementation of PCA is that it requires the whole training set to fit in @@ -1432,7 +1470,7 @@ instances arrive).

    -

    Randomized PCA

    +

    Randomized PCA

    Scikit-Learn offers yet another option to perform PCA, called Randomized PCA. This is a stochastic @@ -1446,7 +1484,7 @@ previous algorithms when \( d \) is much smaller than \( n \).

    -

    Kernel PCA

    +

    Kernel PCA

    @@ -1472,7 +1510,7 @@ X_reduced = rbf_pca.fit_transform(X)

    -

    LLE

    +

    LLE

    Locally Linear Embedding (LLE) is another very powerful nonlinear dimensionality reduction @@ -1484,7 +1522,7 @@ these local relationships are best preserved (more details shortly).

    -

    Other techniques

    +

    Other techniques

    There are many other dimensionality reduction techniques, several of which are available in Scikit-Learn. diff --git a/doc/pub/DimRed/html/DimRed-solarized.html b/doc/pub/DimRed/html/DimRed-solarized.html index a70ae5c1a..af3fb6759 100644 --- a/doc/pub/DimRed/html/DimRed-solarized.html +++ b/doc/pub/DimRed/html/DimRed-solarized.html @@ -124,15 +124,20 @@ div { text-align: justify; text-justify: inter-word; } ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -174,7 +179,7 @@ MathJax.Hub.Config({

    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Dec 28, 2019

    +

    Dec 29, 2019












    @@ -541,6 +546,15 @@ applications.

    Basic ideas of the Principal Component Analysis (PCA)

    +

    +The principal component analysis deals with the problem of fitting a +low-dimensional affine subspace \( S \) of dimension \( d \) much smaller than +the totaldimension \( D \) of the problem at hand (our data +set). Mathematically it can be formulated as a statistical problem or +a geometric problem. In our discussion of the theorem for the +classical PCA, we will stay with a statistical approach. This is also +what set the scene historically which for the PCA. +

    We have a data set defined by a design/feature matrix \( \boldsymbol{X} \) (see below for its definition) @@ -556,7 +570,7 @@ We have a data set defined by a design/feature matrix \( \boldsymbol{X} \) (see

    Before we discuss the PCA theorem, we need to remind ourselves about -the definition of the covariance and the correlation function. +the definition of the covariance and the correlation function. These are quantities

    Suppose we have defined two vectors @@ -616,7 +630,9 @@ In the above example this is the function we constructed using pandas.

    Correlation Function and Design/Feature Matrix

    -In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix \( \boldsymbol{X} \) as +In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression +we defined the design/feature matrix \( \boldsymbol{X} \) as + $$ \boldsymbol{X}=\begin{bmatrix} x_{0,0} & x_{0,1} & x_{0,2}& \dots & \dots x_{0,p-1}\\ @@ -641,7 +657,11 @@ $$ $$

    -With these definitions, we can now rewrite our \( 2\times 2 \) correaltion/covariance matrix in terms of a moe general design/feature matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i =0,1,\dots,p-1 \) +With these definitions, we can now rewrite our \( 2\times 2 \) +correaltion/covariance matrix in terms of a moe general design/feature +matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) +covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i=0,1,\dots,p-1 \) + $$ \boldsymbol{C}[\boldsymbol{x}] = \begin{bmatrix} \mathrm{var}[\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_1] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_2] & \dots & \dots & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_{p-1}]\\ @@ -840,8 +860,8 @@ columns since all matrix elements in the design matrix were set to one

    This means that the variance for these elements will be zero and will cause problems when we set up the correlation matrix. We can simply -drop these elements as follows and then construct the correlation -matrix. +drop these elements and construct a correlation +matrix without these elements.











    @@ -1001,7 +1021,7 @@ X = np.random.multivariate_normal(mean, cov, n) Make thereafter a small Python code which plots the data. Note that the function multivariate returns also the covariance discussed above and that it is defined by dividing by \( n-1 \) instead of \( n \).

    -Now we are going to implement the PCA algorithm. We will break it down into sub-steps and across multiple cells. +Now we are going to implement the PCA algorithm. We will break it down into various substeps.

    Compute the sample mean and center the data

    @@ -1071,8 +1091,11 @@ Finally, try out your own PCA function with other data sets.

    Classical PCA Theorem

    -We assume now that we have a design matrix \( \boldsymbol{X} \) which has been centered as discussed above. For the sake of simplicity we skip the overline symbol. The matrix is defined in terms of the various column vectors \( [\boldsymbol{x}_0,\boldsymbol{x}_1,\dots, \boldsymbol{x}_{p-1}] \) -each with dimension \( \boldsymbol{x}\in {\mathbb{R}}^{n} \). +We assume now that we have a design matrix \( \boldsymbol{X} \) which has been +centered as discussed above. For the sake of simplicity we skip the +overline symbol. The matrix is defined in terms of the various column +vectors \( [\boldsymbol{x}_0,\boldsymbol{x}_1,\dots, \boldsymbol{x}_{p-1}] \) each with dimension +\( \boldsymbol{x}\in {\mathbb{R}}^{n} \).

    We assume also that we have an orthogonal transformation \( \boldsymbol{W}\in {\mathbb{R}}^{p\times p} \). We define the reconstruction error (which is similar to the mean squared error we have seen before) as @@ -1084,7 +1107,13 @@ with \( \overline{\boldsymbol{x}}_i = \boldsymbol{W}\boldsymbol{z}_i \), where \ \( \boldsymbol{Z}\in{\mathbb{R}}^{p\times n} \). When doing PCA we want to reduce this dimensionality.

    -The PCA theorem states that minimizing the above reconstruction error corresponds to setting \( \boldsymbol{W}=\boldsymbol{S} \), the orthogonal matrix which diagonalizes the empirical covariance(correlation) matrix. The optimal low-dimensional encoding of the data is then given by a set of vectors \( \boldsymbol{z}_i \) with at most \( l \) vectors, with \( l < < p \), defined by the orthogonal projection of the data onto the columns spanned by the eigenvectors of the covariance(correlations matrix). +The PCA theorem states that minimizing the above reconstruction error +corresponds to setting \( \boldsymbol{W}=\boldsymbol{S} \), the orthogonal matrix which +diagonalizes the empirical covariance(correlation) matrix. The optimal +low-dimensional encoding of the data is then given by a set of vectors +\( \boldsymbol{z}_i \) with at most \( l \) vectors, with \( l < < p \), defined by the +orthogonal projection of the data onto the columns spanned by the +eigenvectors of the covariance(correlations matrix).











    @@ -1145,7 +1174,10 @@ $$ $$

    -We are almost there, we have obtained a relation between minimizing the reconstruction error and the variance and the covariance matrix. Minimizing the error is equivalent to maximizing the variance of the projected data. +We are almost there, we have obtained a relation between minimizing +the reconstruction error and the variance and the covariance +matrix. Minimizing the error is equivalent to maximizing the variance +of the projected data.











    @@ -1198,10 +1230,21 @@ discussion in chapter 12.2 of Murphy's text has also a nice link with the Singular Value Decomposition theorem. For categorical data, see chapter 12.4 and discussion therein. +

    +Additional part of the proof for the other eigenvectors will be added by mid January 2020. +











    -

    Principal Component Analysis

    +

    Geometric Interpretation and link with Singular Value Decomposition

    + +

    +This material will be added by mid January 2020. + +

    +









    + +

    Principal Component Analysis

    @@ -1257,7 +1300,7 @@ X2D = X_centered.dot(W2)

    -

    PCA and scikit-learn

    +

    PCA and scikit-learn

    Scikit-Learn’s PCA class implements PCA using SVD decomposition just like we did before. The @@ -1289,7 +1332,7 @@ variance that lies along the axis of each principal component.











    -

    Back to the Cancer Data

    +

    Back to the Cancer Data

    We can now repeat the above but applied to real data, in this case our breast cancer data. Here we compute performance scores on the training data using logistic regression.

    @@ -1330,7 +1373,7 @@ We see that our training data after the PCA decomposition has a performance simi











    -

    More on the PCA

    +

    More on the PCA

    Instead of arbitrarily choosing the number of dimensions to reduce down to, it is generally preferable to @@ -1360,7 +1403,7 @@ X_reduced = pca.fit_transform(X)











    -

    Incremental PCA

    +

    Incremental PCA

    One problem with the preceding implementation of PCA is that it requires the whole training set to fit in @@ -1372,7 +1415,7 @@ instances arrive).











    -

    Randomized PCA

    +

    Randomized PCA

    Scikit-Learn offers yet another option to perform PCA, called Randomized PCA. This is a stochastic @@ -1387,7 +1430,7 @@ previous algorithms when \( d \) is much smaller than \( n \).











    -

    Kernel PCA

    +

    Kernel PCA

    @@ -1416,7 +1459,7 @@ X_reduced = rbf_pca.fit_transform(X)











    -

    LLE

    +

    LLE

    Locally Linear Embedding (LLE) is another very powerful nonlinear dimensionality reduction @@ -1428,7 +1471,7 @@ these local relationships are best preserved (more details shortly).











    -

    Other techniques

    +

    Other techniques

    There are many other dimensionality reduction techniques, several of which are available in Scikit-Learn. diff --git a/doc/pub/DimRed/html/DimRed.html b/doc/pub/DimRed/html/DimRed.html index d300d5422..e9a79a081 100644 --- a/doc/pub/DimRed/html/DimRed.html +++ b/doc/pub/DimRed/html/DimRed.html @@ -129,15 +129,20 @@ div { text-align: justify; text-justify: inter-word; } ('Proof of the PCA Theorem', 2, None, '___sec22'), ('PCA Proof continued', 2, None, '___sec23'), ('The final step', 2, None, '___sec24'), - ('Principal Component Analysis', 2, None, '___sec25'), - ('PCA and scikit-learn', 2, None, '___sec26'), - ('Back to the Cancer Data', 2, None, '___sec27'), - ('More on the PCA', 2, None, '___sec28'), - ('Incremental PCA', 2, None, '___sec29'), - ('Randomized PCA', 2, None, '___sec30'), - ('Kernel PCA', 2, None, '___sec31'), - ('LLE', 2, None, '___sec32'), - ('Other techniques', 2, None, '___sec33')]} + ('Geometric Interpretation and link with Singular Value ' + 'Decomposition', + 2, + None, + '___sec25'), + ('Principal Component Analysis', 2, None, '___sec26'), + ('PCA and scikit-learn', 2, None, '___sec27'), + ('Back to the Cancer Data', 2, None, '___sec28'), + ('More on the PCA', 2, None, '___sec29'), + ('Incremental PCA', 2, None, '___sec30'), + ('Randomized PCA', 2, None, '___sec31'), + ('Kernel PCA', 2, None, '___sec32'), + ('LLE', 2, None, '___sec33'), + ('Other techniques', 2, None, '___sec34')]} end of tocinfo --> @@ -179,7 +184,7 @@ MathJax.Hub.Config({

    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Dec 28, 2019

    +

    Dec 29, 2019












    @@ -546,6 +551,15 @@ applications.

    Basic ideas of the Principal Component Analysis (PCA)

    +

    +The principal component analysis deals with the problem of fitting a +low-dimensional affine subspace \( S \) of dimension \( d \) much smaller than +the totaldimension \( D \) of the problem at hand (our data +set). Mathematically it can be formulated as a statistical problem or +a geometric problem. In our discussion of the theorem for the +classical PCA, we will stay with a statistical approach. This is also +what set the scene historically which for the PCA. +

    We have a data set defined by a design/feature matrix \( \boldsymbol{X} \) (see below for its definition) @@ -561,7 +575,7 @@ We have a data set defined by a design/feature matrix \( \boldsymbol{X} \) (see

    Before we discuss the PCA theorem, we need to remind ourselves about -the definition of the covariance and the correlation function. +the definition of the covariance and the correlation function. These are quantities

    Suppose we have defined two vectors @@ -621,7 +635,9 @@ In the above example this is the function we constructed using pandas.

    Correlation Function and Design/Feature Matrix

    -In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix \( \boldsymbol{X} \) as +In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression +we defined the design/feature matrix \( \boldsymbol{X} \) as + $$ \boldsymbol{X}=\begin{bmatrix} x_{0,0} & x_{0,1} & x_{0,2}& \dots & \dots x_{0,p-1}\\ @@ -646,7 +662,11 @@ $$ $$

    -With these definitions, we can now rewrite our \( 2\times 2 \) correaltion/covariance matrix in terms of a moe general design/feature matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i =0,1,\dots,p-1 \) +With these definitions, we can now rewrite our \( 2\times 2 \) +correaltion/covariance matrix in terms of a moe general design/feature +matrix \( \boldsymbol{X}\in {\mathbb{R}}^{n\times p} \). This leads to a \( p\times p \) +covariance matrix for the vectors \( \boldsymbol{x}_i \) with \( i=0,1,\dots,p-1 \) + $$ \boldsymbol{C}[\boldsymbol{x}] = \begin{bmatrix} \mathrm{var}[\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_1] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_2] & \dots & \dots & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_{p-1}]\\ @@ -845,8 +865,8 @@ columns since all matrix elements in the design matrix were set to one

    This means that the variance for these elements will be zero and will cause problems when we set up the correlation matrix. We can simply -drop these elements as follows and then construct the correlation -matrix. +drop these elements and construct a correlation +matrix without these elements.











    @@ -1006,7 +1026,7 @@ X = np.r Make thereafter a small Python code which plots the data. Note that the function multivariate returns also the covariance discussed above and that it is defined by dividing by \( n-1 \) instead of \( n \).

    -Now we are going to implement the PCA algorithm. We will break it down into sub-steps and across multiple cells. +Now we are going to implement the PCA algorithm. We will break it down into various substeps.

    Compute the sample mean and center the data

    @@ -1076,8 +1096,11 @@ Finally, try out your own PCA function with other data sets.

    Classical PCA Theorem

    -We assume now that we have a design matrix \( \boldsymbol{X} \) which has been centered as discussed above. For the sake of simplicity we skip the overline symbol. The matrix is defined in terms of the various column vectors \( [\boldsymbol{x}_0,\boldsymbol{x}_1,\dots, \boldsymbol{x}_{p-1}] \) -each with dimension \( \boldsymbol{x}\in {\mathbb{R}}^{n} \). +We assume now that we have a design matrix \( \boldsymbol{X} \) which has been +centered as discussed above. For the sake of simplicity we skip the +overline symbol. The matrix is defined in terms of the various column +vectors \( [\boldsymbol{x}_0,\boldsymbol{x}_1,\dots, \boldsymbol{x}_{p-1}] \) each with dimension +\( \boldsymbol{x}\in {\mathbb{R}}^{n} \).

    We assume also that we have an orthogonal transformation \( \boldsymbol{W}\in {\mathbb{R}}^{p\times p} \). We define the reconstruction error (which is similar to the mean squared error we have seen before) as @@ -1089,7 +1112,13 @@ with \( \overline{\boldsymbol{x}}_i = \boldsymbol{W}\boldsymbol{z}_i \), where \ \( \boldsymbol{Z}\in{\mathbb{R}}^{p\times n} \). When doing PCA we want to reduce this dimensionality.

    -The PCA theorem states that minimizing the above reconstruction error corresponds to setting \( \boldsymbol{W}=\boldsymbol{S} \), the orthogonal matrix which diagonalizes the empirical covariance(correlation) matrix. The optimal low-dimensional encoding of the data is then given by a set of vectors \( \boldsymbol{z}_i \) with at most \( l \) vectors, with \( l < < p \), defined by the orthogonal projection of the data onto the columns spanned by the eigenvectors of the covariance(correlations matrix). +The PCA theorem states that minimizing the above reconstruction error +corresponds to setting \( \boldsymbol{W}=\boldsymbol{S} \), the orthogonal matrix which +diagonalizes the empirical covariance(correlation) matrix. The optimal +low-dimensional encoding of the data is then given by a set of vectors +\( \boldsymbol{z}_i \) with at most \( l \) vectors, with \( l < < p \), defined by the +orthogonal projection of the data onto the columns spanned by the +eigenvectors of the covariance(correlations matrix).











    @@ -1150,7 +1179,10 @@ $$ $$

    -We are almost there, we have obtained a relation between minimizing the reconstruction error and the variance and the covariance matrix. Minimizing the error is equivalent to maximizing the variance of the projected data. +We are almost there, we have obtained a relation between minimizing +the reconstruction error and the variance and the covariance +matrix. Minimizing the error is equivalent to maximizing the variance +of the projected data.











    @@ -1203,10 +1235,21 @@ discussion in chapter 12.2 of Murphy's text has also a nice link with the Singular Value Decomposition theorem. For categorical data, see chapter 12.4 and discussion therein. +

    +Additional part of the proof for the other eigenvectors will be added by mid January 2020. +











    -

    Principal Component Analysis

    +

    Geometric Interpretation and link with Singular Value Decomposition

    + +

    +This material will be added by mid January 2020. + +

    +









    + +

    Principal Component Analysis

    @@ -1262,7 +1305,7 @@ X2D = X_centered -

    PCA and scikit-learn

    +

    PCA and scikit-learn

    Scikit-Learn’s PCA class implements PCA using SVD decomposition just like we did before. The @@ -1294,7 +1337,7 @@ variance that lies along the axis of each principal component.











    -

    Back to the Cancer Data

    +

    Back to the Cancer Data

    We can now repeat the above but applied to real data, in this case our breast cancer data. Here we compute performance scores on the training data using logistic regression.

    @@ -1335,7 +1378,7 @@ We see that our training data after the PCA decomposition has a performance simi











    -

    More on the PCA

    +

    More on the PCA

    Instead of arbitrarily choosing the number of dimensions to reduce down to, it is generally preferable to @@ -1365,7 +1408,7 @@ X_reduced = pca











    -

    Incremental PCA

    +

    Incremental PCA

    One problem with the preceding implementation of PCA is that it requires the whole training set to fit in @@ -1377,7 +1420,7 @@ instances arrive).











    -

    Randomized PCA

    +

    Randomized PCA

    Scikit-Learn offers yet another option to perform PCA, called Randomized PCA. This is a stochastic @@ -1392,7 +1435,7 @@ previous algorithms when \( d \) is much smaller than \( n \).











    -

    Kernel PCA

    +

    Kernel PCA

    @@ -1421,7 +1464,7 @@ X_reduced = rbf_pcaLLE +

    LLE

    Locally Linear Embedding (LLE) is another very powerful nonlinear dimensionality reduction @@ -1433,7 +1476,7 @@ these local relationships are best preserved (more details shortly).











    -

    Other techniques

    +

    Other techniques

    There are many other dimensionality reduction techniques, several of which are available in Scikit-Learn. diff --git a/doc/pub/DimRed/ipynb/DimRed.ipynb b/doc/pub/DimRed/ipynb/DimRed.ipynb index f688683cf..ec03b06a5 100644 --- a/doc/pub/DimRed/ipynb/DimRed.ipynb +++ b/doc/pub/DimRed/ipynb/DimRed.ipynb @@ -10,7 +10,7 @@ " \n", "**Morten Hjorth-Jensen**, Department of Physics, University of Oslo and Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University\n", "\n", - "Date: **Dec 28, 2019**\n", + "Date: **Dec 29, 2019**\n", "\n", "Copyright 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license\n", "\n", @@ -409,6 +409,14 @@ "\n", "## Basic ideas of the Principal Component Analysis (PCA)\n", "\n", + "The principal component analysis deals with the problem of fitting a\n", + "low-dimensional affine subspace $S$ of dimension $d$ much smaller than\n", + "the totaldimension $D$ of the problem at hand (our data\n", + "set). Mathematically it can be formulated as a statistical problem or\n", + "a geometric problem. In our discussion of the theorem for the\n", + "classical PCA, we will stay with a statistical approach. This is also\n", + "what set the scene historically which for the PCA.\n", + "\n", "We have a data set defined by a design/feature matrix $\\boldsymbol{X}$ (see below for its definition) \n", "* Each data point is determined by $p$ extrinsic (measurement) variables\n", "\n", @@ -419,7 +427,7 @@ "## Introducing the Covariance and Correlation functions\n", "\n", "Before we discuss the PCA theorem, we need to remind ourselves about\n", - "the definition of the covariance and the correlation function.\n", + "the definition of the covariance and the correlation function. These are quantities \n", "\n", "Suppose we have defined two vectors\n", "$\\hat{x}$ and $\\hat{y}$ with $n$ elements each. The covariance matrix $\\boldsymbol{C}$ is defined as" @@ -535,7 +543,8 @@ "\n", "## Correlation Function and Design/Feature Matrix\n", "\n", - "In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix $\\boldsymbol{X}$ as" + "In our derivation of the various regression algorithms like **Ordinary Least Squares** or **Ridge regression**\n", + "we defined the design/feature matrix $\\boldsymbol{X}$ as" ] }, { @@ -592,7 +601,10 @@ "cell_type": "markdown", "metadata": {}, "source": [ - "With these definitions, we can now rewrite our $2\\times 2$ correaltion/covariance matrix in terms of a moe general design/feature matrix $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$. This leads to a $p\\times p$ covariance matrix for the vectors $\\boldsymbol{x}_i$ with $i =0,1,\\dots,p-1$" + "With these definitions, we can now rewrite our $2\\times 2$\n", + "correaltion/covariance matrix in terms of a moe general design/feature\n", + "matrix $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$. This leads to a $p\\times p$\n", + "covariance matrix for the vectors $\\boldsymbol{x}_i$ with $i=0,1,\\dots,p-1$" ] }, { @@ -848,8 +860,8 @@ "\n", "This means that the variance for these elements will be zero and will\n", "cause problems when we set up the correlation matrix. We can simply\n", - "drop these elements as follows and then construct the correlation\n", - "matrix. \n", + "drop these elements and construct a correlation\n", + "matrix without these elements. \n", "\n", "\n", "## Rewriting the Covariance and/or Correlation Matrix\n", @@ -1104,7 +1116,7 @@ "source": [ "Make thereafter a small Python code which plots the data. Note that the function **multivariate** returns also the covariance discussed above and that it is defined by dividing by $n-1$ instead of $n$.\n", "\n", - "Now we are going to implement the PCA algorithm. We will break it down into sub-steps and across multiple cells.\n", + "Now we are going to implement the PCA algorithm. We will break it down into various substeps.\n", "\n", "### Compute the sample mean and center the data\n", "\n", @@ -1208,8 +1220,11 @@ "\n", "## Classical PCA Theorem\n", "\n", - "We assume now that we have a design matrix $\\boldsymbol{X}$ which has been centered as discussed above. For the sake of simplicity we skip the overline symbol. The matrix is defined in terms of the various column vectors $[\\boldsymbol{x}_0,\\boldsymbol{x}_1,\\dots, \\boldsymbol{x}_{p-1}]$\n", - "each with dimension $\\boldsymbol{x}\\in {\\mathbb{R}}^{n}$.\n", + "We assume now that we have a design matrix $\\boldsymbol{X}$ which has been\n", + "centered as discussed above. For the sake of simplicity we skip the\n", + "overline symbol. The matrix is defined in terms of the various column\n", + "vectors $[\\boldsymbol{x}_0,\\boldsymbol{x}_1,\\dots, \\boldsymbol{x}_{p-1}]$ each with dimension\n", + "$\\boldsymbol{x}\\in {\\mathbb{R}}^{n}$.\n", "\n", "We assume also that we have an orthogonal transformation $\\boldsymbol{W}\\in {\\mathbb{R}}^{p\\times p}$. We define the reconstruction error (which is similar to the mean squared error we have seen before) as" ] @@ -1230,7 +1245,13 @@ "with $\\overline{\\boldsymbol{x}}_i = \\boldsymbol{W}\\boldsymbol{z}_i$, where $\\boldsymbol{z}_i$ is a row vector with dimension ${\\mathbb{R}}^{n}$ of the matrix\n", "$\\boldsymbol{Z}\\in{\\mathbb{R}}^{p\\times n}$. When doing PCA we want to reduce this dimensionality. \n", "\n", - "The PCA theorem states that minimizing the above reconstruction error corresponds to setting $\\boldsymbol{W}=\\boldsymbol{S}$, the orthogonal matrix which diagonalizes the empirical covariance(correlation) matrix. The optimal low-dimensional encoding of the data is then given by a set of vectors $\\boldsymbol{z}_i$ with at most $l$ vectors, with $l << p$, defined by the orthogonal projection of the data onto the columns spanned by the eigenvectors of the covariance(correlations matrix).\n", + "The PCA theorem states that minimizing the above reconstruction error\n", + "corresponds to setting $\\boldsymbol{W}=\\boldsymbol{S}$, the orthogonal matrix which\n", + "diagonalizes the empirical covariance(correlation) matrix. The optimal\n", + "low-dimensional encoding of the data is then given by a set of vectors\n", + "$\\boldsymbol{z}_i$ with at most $l$ vectors, with $l << p$, defined by the\n", + "orthogonal projection of the data onto the columns spanned by the\n", + "eigenvectors of the covariance(correlations matrix).\n", "\n", "\n", "\n", @@ -1371,7 +1392,10 @@ "cell_type": "markdown", "metadata": {}, "source": [ - "We are almost there, we have obtained a relation between minimizing the reconstruction error and the variance and the covariance matrix. Minimizing the error is equivalent to maximizing the variance of the projected data. \n", + "We are almost there, we have obtained a relation between minimizing\n", + "the reconstruction error and the variance and the covariance\n", + "matrix. Minimizing the error is equivalent to maximizing the variance\n", + "of the projected data.\n", "\n", "## The final step\n", "\n", @@ -1460,9 +1484,11 @@ "the Singular Value Decomposition theorem. For categorical data, see\n", "chapter 12.4 and discussion therein.\n", "\n", + "Additional part of the proof for the other eigenvectors will be added by mid January 2020.\n", "\n", + "## Geometric Interpretation and link with Singular Value Decomposition\n", "\n", - "\n", + "This material will be added by mid January 2020.\n", "\n", "\n", "## Principal Component Analysis\n", diff --git a/doc/pub/DimRed/ipynb/ipynb-DimRed-src.tar.gz b/doc/pub/DimRed/ipynb/ipynb-DimRed-src.tar.gz index 2d4521bc7..363947881 100644 Binary files a/doc/pub/DimRed/ipynb/ipynb-DimRed-src.tar.gz and b/doc/pub/DimRed/ipynb/ipynb-DimRed-src.tar.gz differ diff --git a/doc/pub/DimRed/pdf/DimRed-minted.pdf b/doc/pub/DimRed/pdf/DimRed-minted.pdf index 4e25cee27..a72e6b2de 100644 Binary files a/doc/pub/DimRed/pdf/DimRed-minted.pdf and b/doc/pub/DimRed/pdf/DimRed-minted.pdf differ diff --git a/doc/src/DimRed/DimRed.do.txt b/doc/src/DimRed/DimRed.do.txt index 077e1be6c..621e66422 100644 --- a/doc/src/DimRed/DimRed.do.txt +++ b/doc/src/DimRed/DimRed.do.txt @@ -337,6 +337,14 @@ applications. !split ===== Basic ideas of the Principal Component Analysis (PCA) ===== +The principal component analysis deals with the problem of fitting a +low-dimensional affine subspace $S$ of dimension $d$ much smaller than +the totaldimension $D$ of the problem at hand (our data +set). Mathematically it can be formulated as a statistical problem or +a geometric problem. In our discussion of the theorem for the +classical PCA, we will stay with a statistical approach. This is also +what set the scene historically which for the PCA. + We have a data set defined by a design/feature matrix $\bm{X}$ (see below for its definition) * Each data point is determined by $p$ extrinsic (measurement) variables * We may want to ask the following question: Are there fewer intrinsic variables (say $d << p$) that still approximately describe the data? @@ -347,7 +355,7 @@ We have a data set defined by a design/feature matrix $\bm{X}$ (see below for it ===== Introducing the Covariance and Correlation functions ===== Before we discuss the PCA theorem, we need to remind ourselves about -the definition of the covariance and the correlation function. +the definition of the covariance and the correlation function. These are quantities Suppose we have defined two vectors $\hat{x}$ and $\hat{y}$ with $n$ elements each. The covariance matrix $\bm{C}$ is defined as @@ -409,7 +417,9 @@ In the above example this is the function we constructed using _pandas_. !split ===== Correlation Function and Design/Feature Matrix ===== -In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression we defined the design/feature matrix $\bm{X}$ as +In our derivation of the various regression algorithms like _Ordinary Least Squares_ or _Ridge regression_ +we defined the design/feature matrix $\bm{X}$ as + !bt \[ \bm{X}=\begin{bmatrix} @@ -437,7 +447,11 @@ with a given vector \] !et -With these definitions, we can now rewrite our $2\times 2$ correaltion/covariance matrix in terms of a moe general design/feature matrix $\bm{X}\in {\mathbb{R}}^{n\times p}$. This leads to a $p\times p$ covariance matrix for the vectors $\bm{x}_i$ with $i =0,1,\dots,p-1$ +With these definitions, we can now rewrite our $2\times 2$ +correaltion/covariance matrix in terms of a moe general design/feature +matrix $\bm{X}\in {\mathbb{R}}^{n\times p}$. This leads to a $p\times p$ +covariance matrix for the vectors $\bm{x}_i$ with $i=0,1,\dots,p-1$ + !bt \[ \bm{C}[\bm{x}] = \begin{bmatrix} @@ -624,8 +638,8 @@ columns since all matrix elements in the design matrix were set to one This means that the variance for these elements will be zero and will cause problems when we set up the correlation matrix. We can simply -drop these elements as follows and then construct the correlation -matrix. +drop these elements and construct a correlation +matrix without these elements. !split @@ -773,7 +787,7 @@ X = np.random.multivariate_normal(mean, cov, n) Make thereafter a small Python code which plots the data. Note that the function _multivariate_ returns also the covariance discussed above and that it is defined by dividing by $n-1$ instead of $n$. -Now we are going to implement the PCA algorithm. We will break it down into sub-steps and across multiple cells. +Now we are going to implement the PCA algorithm. We will break it down into various substeps. === Compute the sample mean and center the data === @@ -835,8 +849,11 @@ Finally, try out your own PCA function with other data sets. !split ===== Classical PCA Theorem ===== -We assume now that we have a design matrix $\bm{X}$ which has been centered as discussed above. For the sake of simplicity we skip the overline symbol. The matrix is defined in terms of the various column vectors $[\bm{x}_0,\bm{x}_1,\dots, \bm{x}_{p-1}]$ -each with dimension $\bm{x}\in {\mathbb{R}}^{n}$. +We assume now that we have a design matrix $\bm{X}$ which has been +centered as discussed above. For the sake of simplicity we skip the +overline symbol. The matrix is defined in terms of the various column +vectors $[\bm{x}_0,\bm{x}_1,\dots, \bm{x}_{p-1}]$ each with dimension +$\bm{x}\in {\mathbb{R}}^{n}$. We assume also that we have an orthogonal transformation $\bm{W}\in {\mathbb{R}}^{p\times p}$. We define the reconstruction error (which is similar to the mean squared error we have seen before) as !bt @@ -847,7 +864,13 @@ J(\bm{W},\bm{Z}) = \frac{1}{n}\sum_i (\bm{x}_i - \overline{\bm{x}}_i)^2, with $\overline{\bm{x}}_i = \bm{W}\bm{z}_i$, where $\bm{z}_i$ is a row vector with dimension ${\mathbb{R}}^{n}$ of the matrix $\bm{Z}\in{\mathbb{R}}^{p\times n}$. When doing PCA we want to reduce this dimensionality. -The PCA theorem states that minimizing the above reconstruction error corresponds to setting $\bm{W}=\bm{S}$, the orthogonal matrix which diagonalizes the empirical covariance(correlation) matrix. The optimal low-dimensional encoding of the data is then given by a set of vectors $\bm{z}_i$ with at most $l$ vectors, with $l << p$, defined by the orthogonal projection of the data onto the columns spanned by the eigenvectors of the covariance(correlations matrix). +The PCA theorem states that minimizing the above reconstruction error +corresponds to setting $\bm{W}=\bm{S}$, the orthogonal matrix which +diagonalizes the empirical covariance(correlation) matrix. The optimal +low-dimensional encoding of the data is then given by a set of vectors +$\bm{z}_i$ with at most $l$ vectors, with $l << p$, defined by the +orthogonal projection of the data onto the columns spanned by the +eigenvectors of the covariance(correlations matrix). @@ -912,7 +935,10 @@ we have thus that \] !et -We are almost there, we have obtained a relation between minimizing the reconstruction error and the variance and the covariance matrix. Minimizing the error is equivalent to maximizing the variance of the projected data. +We are almost there, we have obtained a relation between minimizing +the reconstruction error and the variance and the covariance +matrix. Minimizing the error is equivalent to maximizing the variance +of the projected data. !split ===== The final step ===== @@ -965,9 +991,12 @@ discussion in chapter 12.2 of Murphy's text has also a nice link with the Singular Value Decomposition theorem. For categorical data, see chapter 12.4 and discussion therein. +Additional part of the proof for the other eigenvectors will be added by mid January 2020. +!split +===== Geometric Interpretation and link with Singular Value Decomposition ===== - +This material will be added by mid January 2020. !split