433 lines
25 KiB
HTML
433 lines
25 KiB
HTML
<!--
|
|
Automatically generated HTML file from DocOnce source
|
|
(https://github.com/hplgit/doconce/)
|
|
-->
|
|
<html>
|
|
<head>
|
|
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
|
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
|
|
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
<meta name="description" content="Data Analysis and Machine Learning: Preprocessing and Dimensionality Reduction">
|
|
|
|
<title>Data Analysis and Machine Learning: Preprocessing and Dimensionality Reduction</title>
|
|
|
|
<!-- Bootstrap style: bootstrap -->
|
|
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
|
|
<!-- not necessary
|
|
<link href="https://netdna.bootstrapcdn.com/font-awesome/4.0.3/css/font-awesome.css" rel="stylesheet">
|
|
-->
|
|
|
|
<style type="text/css">
|
|
|
|
/* Add scrollbar to dropdown menus in bootstrap navigation bar */
|
|
.dropdown-menu {
|
|
height: auto;
|
|
max-height: 400px;
|
|
overflow-x: hidden;
|
|
}
|
|
|
|
/* Adds an invisible element before each target to offset for the navigation
|
|
bar */
|
|
.anchor::before {
|
|
content:"";
|
|
display:block;
|
|
height:50px; /* fixed header height for style bootstrap */
|
|
margin:-50px 0 0; /* negative fixed header height */
|
|
}
|
|
</style>
|
|
|
|
|
|
</head>
|
|
|
|
<!-- tocinfo
|
|
{'highest level': 2,
|
|
'sections': [('Reducing the number of degrees of freedom, overarching view',
|
|
2,
|
|
None,
|
|
'___sec0'),
|
|
('Preprocessing our data', 2, None, '___sec1'),
|
|
('More preprocessing', 2, None, '___sec2'),
|
|
('Simple preprocessing examples, Franke function and regression',
|
|
2,
|
|
None,
|
|
'___sec3'),
|
|
('Simple preprocessing examples, breast cancer data and '
|
|
'classification, Support Vector Machines',
|
|
2,
|
|
None,
|
|
'___sec4'),
|
|
('More on Cancer Data, now with Logistic Regression',
|
|
2,
|
|
None,
|
|
'___sec5'),
|
|
('Why should we think of reducing the dimensionality',
|
|
2,
|
|
None,
|
|
'___sec6'),
|
|
('Basic ideas of the Principal Component Analysis (PCA)',
|
|
2,
|
|
None,
|
|
'___sec7'),
|
|
('Introducing the Covariance and Correlation functions',
|
|
2,
|
|
None,
|
|
'___sec8'),
|
|
('Correlation Function and Design/Feature Matrix',
|
|
2,
|
|
None,
|
|
'___sec9'),
|
|
('Covariance Matrix Examples', 2, None, '___sec10'),
|
|
('Correlation Matrix', 2, None, '___sec11'),
|
|
('Correlation Matrix with Pandas', 2, None, '___sec12'),
|
|
('Correlation Matrix with Pandas and the Franke function',
|
|
2,
|
|
None,
|
|
'___sec13'),
|
|
('Rewriting the Covariance and/or Correlation Matrix',
|
|
2,
|
|
None,
|
|
'___sec14'),
|
|
('Towards the PCA theorem', 2, None, '___sec15'),
|
|
('The Algorithm before the Theorem', 2, None, '___sec16'),
|
|
('Writing our own PCA code', 2, None, '___sec17'),
|
|
('Compute the sample mean and center the data',
|
|
3,
|
|
None,
|
|
'___sec18'),
|
|
('Compute the sample covariance', 3, None, '___sec19'),
|
|
('Diagonalize the sample covariance matrix to obtain the '
|
|
'principal components',
|
|
3,
|
|
None,
|
|
'___sec20'),
|
|
('Classical PCA Theorem', 2, None, '___sec21'),
|
|
('Proof of the PCA Theorem', 2, None, '___sec22'),
|
|
('PCA Proof continued', 2, None, '___sec23'),
|
|
('The final step', 2, None, '___sec24'),
|
|
('Geometric Interpretation and link with Singular Value '
|
|
'Decomposition',
|
|
2,
|
|
None,
|
|
'___sec25'),
|
|
('Principal Component Analysis', 2, None, '___sec26'),
|
|
('PCA and scikit-learn', 2, None, '___sec27'),
|
|
('Back to the Cancer Data', 2, None, '___sec28'),
|
|
('More on the PCA', 2, None, '___sec29'),
|
|
('Incremental PCA', 2, None, '___sec30'),
|
|
('Randomized PCA', 2, None, '___sec31'),
|
|
('Kernel PCA', 2, None, '___sec32'),
|
|
('LLE', 2, None, '___sec33'),
|
|
('Other techniques', 2, None, '___sec34')]}
|
|
end of tocinfo -->
|
|
|
|
<body>
|
|
|
|
|
|
|
|
<script type="text/x-mathjax-config">
|
|
MathJax.Hub.Config({
|
|
TeX: {
|
|
equationNumbers: { autoNumber: "none" },
|
|
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
|
|
}
|
|
});
|
|
</script>
|
|
<script type="text/javascript" async
|
|
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
|
|
</script>
|
|
|
|
|
|
|
|
|
|
<!-- Bootstrap navigation bar -->
|
|
<div class="navbar navbar-default navbar-fixed-top">
|
|
<div class="navbar-header">
|
|
<button type="button" class="navbar-toggle" data-toggle="collapse" data-target=".navbar-responsive-collapse">
|
|
<span class="icon-bar"></span>
|
|
<span class="icon-bar"></span>
|
|
<span class="icon-bar"></span>
|
|
</button>
|
|
<a class="navbar-brand" href="DimRed-bs.html">Data Analysis and Machine Learning: Preprocessing and Dimensionality Reduction</a>
|
|
</div>
|
|
|
|
<div class="navbar-collapse collapse navbar-responsive-collapse">
|
|
<ul class="nav navbar-nav navbar-right">
|
|
<li class="dropdown">
|
|
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
|
|
<ul class="dropdown-menu">
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs001.html#___sec0" style="font-size: 80%;"><b>Reducing the number of degrees of freedom, overarching view</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs002.html#___sec1" style="font-size: 80%;"><b>Preprocessing our data</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs003.html#___sec2" style="font-size: 80%;"><b>More preprocessing</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs004.html#___sec3" style="font-size: 80%;"><b>Simple preprocessing examples, Franke function and regression</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs005.html#___sec4" style="font-size: 80%;"><b>Simple preprocessing examples, breast cancer data and classification, Support Vector Machines</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs006.html#___sec5" style="font-size: 80%;"><b>More on Cancer Data, now with Logistic Regression</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs007.html#___sec6" style="font-size: 80%;"><b>Why should we think of reducing the dimensionality</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs008.html#___sec7" style="font-size: 80%;"><b>Basic ideas of the Principal Component Analysis (PCA)</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs009.html#___sec8" style="font-size: 80%;"><b>Introducing the Covariance and Correlation functions</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs010.html#___sec9" style="font-size: 80%;"><b>Correlation Function and Design/Feature Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs011.html#___sec10" style="font-size: 80%;"><b>Covariance Matrix Examples</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs012.html#___sec11" style="font-size: 80%;"><b>Correlation Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs013.html#___sec12" style="font-size: 80%;"><b>Correlation Matrix with Pandas</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs014.html#___sec13" style="font-size: 80%;"><b>Correlation Matrix with Pandas and the Franke function</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs015.html#___sec14" style="font-size: 80%;"><b>Rewriting the Covariance and/or Correlation Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs016.html#___sec15" style="font-size: 80%;"><b>Towards the PCA theorem</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs017.html#___sec16" style="font-size: 80%;"><b>The Algorithm before the Theorem</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec17" style="font-size: 80%;"><b>Writing our own PCA code</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec18" style="font-size: 80%;"> Compute the sample mean and center the data</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec19" style="font-size: 80%;"> Compute the sample covariance</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec20" style="font-size: 80%;"> Diagonalize the sample covariance matrix to obtain the principal components</a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs019.html#___sec21" style="font-size: 80%;"><b>Classical PCA Theorem</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs020.html#___sec22" style="font-size: 80%;"><b>Proof of the PCA Theorem</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs021.html#___sec23" style="font-size: 80%;"><b>PCA Proof continued</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs022.html#___sec24" style="font-size: 80%;"><b>The final step</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs023.html#___sec25" style="font-size: 80%;"><b>Geometric Interpretation and link with Singular Value Decomposition</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs024.html#___sec26" style="font-size: 80%;"><b>Principal Component Analysis</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs025.html#___sec27" style="font-size: 80%;"><b>PCA and scikit-learn</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs026.html#___sec28" style="font-size: 80%;"><b>Back to the Cancer Data</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs027.html#___sec29" style="font-size: 80%;"><b>More on the PCA</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs028.html#___sec30" style="font-size: 80%;"><b>Incremental PCA</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs029.html#___sec31" style="font-size: 80%;"><b>Randomized PCA</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs030.html#___sec32" style="font-size: 80%;"><b>Kernel PCA</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs031.html#___sec33" style="font-size: 80%;"><b>LLE</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._DimRed-bs032.html#___sec34" style="font-size: 80%;"><b>Other techniques</b></a></li>
|
|
|
|
</ul>
|
|
</li>
|
|
</ul>
|
|
</div>
|
|
</div>
|
|
</div> <!-- end of navigation bar -->
|
|
|
|
<div class="container">
|
|
|
|
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
|
|
|
<a name="part0018"></a>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec17" class="anchor">Writing our own PCA code </h2>
|
|
|
|
<p>
|
|
We will use a simple example first with two-dimensional data
|
|
drawn from a multivariate normal distribution with the following mean and covariance matrix:
|
|
$$
|
|
\mu = (-1,2) \qquad \Sigma = \begin{bmatrix} 4 & 2 \\
|
|
2 & 2
|
|
\end{bmatrix}
|
|
$$
|
|
|
|
Note that the mean refers to each column of data.
|
|
We will generate \( n = 1000 \) points \( X = \{ x_1, \ldots, x_N \} \) from
|
|
this distribution, and store them in the \( 1000 \times 2 \) matrix \( \boldsymbol{X} \).
|
|
|
|
<p>
|
|
The following Python code aids in setting up the data and writing out the design matrix.
|
|
Note that the function <b>multivariate</b> returns also the covariance discussed above and that it is defined by dividing by \( n-1 \) instead of \( n \).
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">IPython.display</span> <span style="color: #008000; font-weight: bold">import</span> display
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">10000</span>
|
|
mean <span style="color: #666666">=</span> (<span style="color: #666666">-1</span>, <span style="color: #666666">2</span>)
|
|
cov <span style="color: #666666">=</span> [[<span style="color: #666666">4</span>, <span style="color: #666666">2</span>], [<span style="color: #666666">2</span>, <span style="color: #666666">2</span>]]
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>multivariate_normal(mean, cov, n)
|
|
</pre></div>
|
|
<p>
|
|
Now we are going to implement the PCA algorithm. We will break it down into various substeps.
|
|
|
|
<h3 id="___sec18" class="anchor">Compute the sample mean and center the data </h3>
|
|
|
|
<p>
|
|
The first step of PCA is to compute the sample mean of the data and use it to center the data. Recall that the sample mean is
|
|
$$
|
|
\mu_n = \frac{1}{n} \sum_{i=1}^n x_i
|
|
$$
|
|
|
|
and the mean-centered data \( \bar{X} = \{ \bar{x}_1, \ldots, \bar{x}_n \} \) takes the form
|
|
$$
|
|
\bar{x}_i = x_i - \mu_n.
|
|
$$
|
|
|
|
When you are done with these steps, print out \( \mu_n \) to verify it is
|
|
close to \( \mu \) and plot your mean centered data to verify it is
|
|
centered at the origin! Compare your code with the functionality from <b>Scikit-Learn</b> discussed above.
|
|
The following code elements perform these operations using <b>pandas</b> or using our own functionality for doing so. The latter, using <b>numpy</b> is rather simple through the <b>mean()</b> function.
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>df <span style="color: #666666">=</span> pd<span style="color: #666666">.</span>DataFrame(X)
|
|
<span style="color: #408080; font-style: italic"># Pandas does the centering for us</span>
|
|
df <span style="color: #666666">=</span> df <span style="color: #666666">-</span>df<span style="color: #666666">.</span>mean()
|
|
<span style="color: #408080; font-style: italic"># we center it ourselves</span>
|
|
X_centered <span style="color: #666666">=</span> X <span style="color: #666666">-</span> X<span style="color: #666666">.</span>mean(axis<span style="color: #666666">=0</span>)
|
|
</pre></div>
|
|
<p>
|
|
Alternatively, we could use the functions we discussed
|
|
earlier for scaling the data set. That is, we could have used the
|
|
<b>StandardScaler</b> function in <b>Scikit-Learn</b>, a function which ensures
|
|
that for each feature/predictor we study the mean value is zero and
|
|
the variance is one (every column in the design/feature matrix). You
|
|
would then not get the same results, since we divide by the
|
|
variance. The diagonal covariance matrix elements will then be one,
|
|
while the non-diagonal ones need to be divided by \( 2\sqrt{2} \) for our
|
|
specific case.
|
|
|
|
<h3 id="___sec19" class="anchor">Compute the sample covariance </h3>
|
|
|
|
<p>
|
|
Now we are going to use the mean centered data to compute the sample covariance of the data by using the following equation
|
|
$$
|
|
\begin{equation*}
|
|
\Sigma_n = \frac{1}{n-1} \sum_{i=1}^n \bar{x}_i^T \bar{x}_i = \frac{1}{n-1} \sum_{i=1}^n (x_i - \mu_n)^T (x_i - \mu_n)
|
|
\end{equation*}
|
|
$$
|
|
|
|
where the data points \( x_i \in \mathbb{R}^p \) (here in this example \( p = 2 \)) are column vectors and \( x^T \) is the transpose of \( x \).
|
|
We can write our own code or simply use either the functionaly of <b>numpy</b> or that of <b>pandas</b>, as follows
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">print</span>(df<span style="color: #666666">.</span>cov())
|
|
<span style="color: #008000; font-weight: bold">print</span>(np<span style="color: #666666">.</span>cov(X_centered<span style="color: #666666">.</span>T))
|
|
</pre></div>
|
|
<p>
|
|
Note that the way we define the covariance matrix here has a factor \( n-1 \) instead of \( n \). This is included in the <b>cov()</b> function by <b>numpy</b> and <b>pandas</b>.
|
|
Our own code here is not very elegant and asks for obvious improvements. It is tailored to this specific \( 2\times 2 \) covariance matrix.
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #408080; font-style: italic"># extract the relevant columns from the centered design matrix of dim n x 2</span>
|
|
x <span style="color: #666666">=</span> X_centered[:,<span style="color: #666666">0</span>]
|
|
y <span style="color: #666666">=</span> X_centered[:,<span style="color: #666666">1</span>]
|
|
Cov <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #666666">2</span>,<span style="color: #666666">2</span>))
|
|
Cov[<span style="color: #666666">0</span>,<span style="color: #666666">1</span>] <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(x<span style="color: #666666">.</span>T<span style="color: #AA22FF">@y</span>)<span style="color: #666666">/</span>(n<span style="color: #666666">-1.0</span>)
|
|
Cov[<span style="color: #666666">0</span>,<span style="color: #666666">0</span>] <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(x<span style="color: #666666">.</span>T<span style="color: #AA22FF">@x</span>)<span style="color: #666666">/</span>(n<span style="color: #666666">-1.0</span>)
|
|
Cov[<span style="color: #666666">1</span>,<span style="color: #666666">1</span>] <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(y<span style="color: #666666">.</span>T<span style="color: #AA22FF">@y</span>)<span style="color: #666666">/</span>(n<span style="color: #666666">-1.0</span>)
|
|
Cov[<span style="color: #666666">1</span>,<span style="color: #666666">0</span>]<span style="color: #666666">=</span> Cov[<span style="color: #666666">0</span>,<span style="color: #666666">1</span>]
|
|
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"Centered covariance using own code"</span>)
|
|
<span style="color: #008000; font-weight: bold">print</span>(Cov)
|
|
plt<span style="color: #666666">.</span>plot(x, y, <span style="color: #BA2121">'x'</span>)
|
|
plt<span style="color: #666666">.</span>axis(<span style="color: #BA2121">'equal'</span>)
|
|
plt<span style="color: #666666">.</span>show()
|
|
</pre></div>
|
|
<p>
|
|
Depending on the number of points \( n \), we will get results that are close to the covariance values defined above.
|
|
The plot shows how the data are clustered around a line with slope close to one. Is this expected?
|
|
|
|
<h3 id="___sec20" class="anchor">Diagonalize the sample covariance matrix to obtain the principal components </h3>
|
|
|
|
<p>
|
|
Now we are ready to solve for the principal components! To do so we
|
|
diagonalize the sample covariance matrix \( \Sigma \). We can use the
|
|
function <b>np.linalg.eig</b> to do so. It will return the eigenvalues and
|
|
eigenvectors of \( \Sigma \). Once we have these we can perform the
|
|
following tasks:
|
|
|
|
<ul>
|
|
<li> We compute the percentage of the total variance captured by the first principal component</li>
|
|
<li> We plot the mean centered data and lines along the first and second principal components</li>
|
|
<li> Then we project the mean centered data onto the first and second principal components, and plot the projected data.</li>
|
|
<li> Finally, we approximate the data as</li>
|
|
</ul>
|
|
|
|
$$
|
|
\begin{equation*}
|
|
x_i \approx \tilde{x}_i = \mu_n + \langle x_i, v_0 \rangle v_0
|
|
\end{equation*}
|
|
$$
|
|
|
|
where \( v_0 \) is the first principal component.
|
|
|
|
<p>
|
|
Collecting all these steps we can write our own PCA function and
|
|
compare this with the functionality included in <b>Scikit-Learn</b>.
|
|
|
|
<p>
|
|
The code here outlines some of the elements we could include in the
|
|
analysis. Feel free to extend upon this in order to address the above
|
|
questions.
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #408080; font-style: italic"># diagonalize and obtain eigenvalues, not necessarily sorted</span>
|
|
EigValues, EigVectors <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>eig(Cov)
|
|
<span style="color: #408080; font-style: italic"># sort eigenvectors and eigenvalues</span>
|
|
<span style="color: #408080; font-style: italic">#permute = EigValues.argsort()</span>
|
|
<span style="color: #408080; font-style: italic">#EigValues = EigValues[permute]</span>
|
|
<span style="color: #408080; font-style: italic">#EigVectors = EigVectors[:,permute]</span>
|
|
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"Eigenvalues of Covariance matrix"</span>)
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">2</span>):
|
|
<span style="color: #008000; font-weight: bold">print</span>(EigValues[i])
|
|
FirstEigvector <span style="color: #666666">=</span> EigVectors[:,<span style="color: #666666">0</span>]
|
|
SecondEigvector <span style="color: #666666">=</span> EigVectors[:,<span style="color: #666666">1</span>]
|
|
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"First eigenvector"</span>)
|
|
<span style="color: #008000; font-weight: bold">print</span>(FirstEigvector)
|
|
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"Second eigenvector"</span>)
|
|
<span style="color: #008000; font-weight: bold">print</span>(SecondEigvector)
|
|
<span style="color: #408080; font-style: italic">#thereafter we do a PCA with Scikit-learn</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.decomposition</span> <span style="color: #008000; font-weight: bold">import</span> PCA
|
|
pca <span style="color: #666666">=</span> PCA(n_components <span style="color: #666666">=</span> <span style="color: #666666">2</span>)
|
|
X2Dsl <span style="color: #666666">=</span> pca<span style="color: #666666">.</span>fit_transform(X)
|
|
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"Eigenvector of largest eigenvalue"</span>)
|
|
<span style="color: #008000; font-weight: bold">print</span>(pca<span style="color: #666666">.</span>components_<span style="color: #666666">.</span>T[:, <span style="color: #666666">0</span>])
|
|
</pre></div>
|
|
<p>
|
|
This code does not contain all the above elements, but it shows how we can use <b>Scikit-Learn</b> to extract the eigenvector which corresponds to the largest eigenvalue. Try to address the questions we pose before the above code. Try also to change the values of the covariance matrix by making one of the diagonal elements much larger than the other. What do you observe then?
|
|
|
|
<p>
|
|
<p>
|
|
<!-- navigation buttons at the bottom of the page -->
|
|
<ul class="pagination">
|
|
<li><a href="._DimRed-bs017.html">«</a></li>
|
|
<li><a href="._DimRed-bs000.html">1</a></li>
|
|
<li><a href="">...</a></li>
|
|
<li><a href="._DimRed-bs010.html">11</a></li>
|
|
<li><a href="._DimRed-bs011.html">12</a></li>
|
|
<li><a href="._DimRed-bs012.html">13</a></li>
|
|
<li><a href="._DimRed-bs013.html">14</a></li>
|
|
<li><a href="._DimRed-bs014.html">15</a></li>
|
|
<li><a href="._DimRed-bs015.html">16</a></li>
|
|
<li><a href="._DimRed-bs016.html">17</a></li>
|
|
<li><a href="._DimRed-bs017.html">18</a></li>
|
|
<li class="active"><a href="._DimRed-bs018.html">19</a></li>
|
|
<li><a href="._DimRed-bs019.html">20</a></li>
|
|
<li><a href="._DimRed-bs020.html">21</a></li>
|
|
<li><a href="._DimRed-bs021.html">22</a></li>
|
|
<li><a href="._DimRed-bs022.html">23</a></li>
|
|
<li><a href="._DimRed-bs023.html">24</a></li>
|
|
<li><a href="._DimRed-bs024.html">25</a></li>
|
|
<li><a href="._DimRed-bs025.html">26</a></li>
|
|
<li><a href="._DimRed-bs026.html">27</a></li>
|
|
<li><a href="._DimRed-bs027.html">28</a></li>
|
|
<li><a href="">...</a></li>
|
|
<li><a href="._DimRed-bs032.html">33</a></li>
|
|
<li><a href="._DimRed-bs019.html">»</a></li>
|
|
</ul>
|
|
<!-- ------------------- end of main content --------------- -->
|
|
|
|
</div> <!-- end container -->
|
|
<!-- include javascript, jQuery *first* -->
|
|
<script src="https://ajax.googleapis.com/ajax/libs/jquery/1.10.2/jquery.min.js"></script>
|
|
<script src="https://netdna.bootstrapcdn.com/bootstrap/3.0.0/js/bootstrap.min.js"></script>
|
|
|
|
<!-- Bootstrap footer
|
|
<footer>
|
|
<a href="http://..."><img width="250" align=right src="http://..."></a>
|
|
</footer>
|
|
-->
|
|
|
|
|
|
<center style="font-size:80%">
|
|
<!-- copyright only on the titlepage -->
|
|
</center>
|
|
|
|
|
|
</body>
|
|
</html>
|
|
|
|
|