Files
FYS-STK4155/doc/pub/DimRed/html/._DimRed-bs018.html
T
2020-01-01 19:57:12 +01:00

405 lines
22 KiB
HTML

<!--
Automatically generated HTML file from DocOnce source
(https://github.com/hplgit/doconce/)
-->
<html>
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<meta name="description" content="Data Analysis and Machine Learning: Preprocessing and Dimensionality Reduction">
<title>Data Analysis and Machine Learning: Preprocessing and Dimensionality Reduction</title>
<!-- Bootstrap style: bootstrap -->
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
<!-- not necessary
<link href="https://netdna.bootstrapcdn.com/font-awesome/4.0.3/css/font-awesome.css" rel="stylesheet">
-->
<style type="text/css">
/* Add scrollbar to dropdown menus in bootstrap navigation bar */
.dropdown-menu {
height: auto;
max-height: 400px;
overflow-x: hidden;
}
/* Adds an invisible element before each target to offset for the navigation
bar */
.anchor::before {
content:"";
display:block;
height:50px; /* fixed header height for style bootstrap */
margin:-50px 0 0; /* negative fixed header height */
}
</style>
</head>
<!-- tocinfo
{'highest level': 2,
'sections': [('Reducing the number of degrees of freedom, overarching view',
2,
None,
'___sec0'),
('Preprocessing our data', 2, None, '___sec1'),
('More preprocessing', 2, None, '___sec2'),
('Simple preprocessing examples, Franke function and regression',
2,
None,
'___sec3'),
('Simple preprocessing examples, breast cancer data and '
'classification, Support Vector Machines',
2,
None,
'___sec4'),
('More on Cancer Data, now with Logistic Regression',
2,
None,
'___sec5'),
('Why should we think of reducing the dimensionality',
2,
None,
'___sec6'),
('Basic ideas of the Principal Component Analysis (PCA)',
2,
None,
'___sec7'),
('Introducing the Covariance and Correlation functions',
2,
None,
'___sec8'),
('Correlation Function and Design/Feature Matrix',
2,
None,
'___sec9'),
('Covariance Matrix Examples', 2, None, '___sec10'),
('Correlation Matrix', 2, None, '___sec11'),
('Correlation Matrix with Pandas', 2, None, '___sec12'),
('Correlation Matrix with Pandas and the Franke function',
2,
None,
'___sec13'),
('Rewriting the Covariance and/or Correlation Matrix',
2,
None,
'___sec14'),
('Towards the PCA theorem', 2, None, '___sec15'),
('The Algorithm before the Theorem', 2, None, '___sec16'),
('Writing our own PCA code', 2, None, '___sec17'),
('Compute the sample mean and center the data',
3,
None,
'___sec18'),
('Compute the sample covariance', 3, None, '___sec19'),
('Diagonalize the sample covariance matrix to obtain the '
'principal components',
3,
None,
'___sec20'),
('Classical PCA Theorem', 2, None, '___sec21'),
('Proof of the PCA Theorem', 2, None, '___sec22'),
('PCA Proof continued', 2, None, '___sec23'),
('The final step', 2, None, '___sec24'),
('Geometric Interpretation and link with Singular Value '
'Decomposition',
2,
None,
'___sec25'),
('Principal Component Analysis', 2, None, '___sec26'),
('PCA and scikit-learn', 2, None, '___sec27'),
('Back to the Cancer Data', 2, None, '___sec28'),
('More on the PCA', 2, None, '___sec29'),
('Incremental PCA', 2, None, '___sec30'),
('Randomized PCA', 2, None, '___sec31'),
('Kernel PCA', 2, None, '___sec32'),
('LLE', 2, None, '___sec33'),
('Other techniques', 2, None, '___sec34')]}
end of tocinfo -->
<body>
<script type="text/x-mathjax-config">
MathJax.Hub.Config({
TeX: {
equationNumbers: { autoNumber: "none" },
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
}
});
</script>
<script type="text/javascript" async
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
</script>
<!-- Bootstrap navigation bar -->
<div class="navbar navbar-default navbar-fixed-top">
<div class="navbar-header">
<button type="button" class="navbar-toggle" data-toggle="collapse" data-target=".navbar-responsive-collapse">
<span class="icon-bar"></span>
<span class="icon-bar"></span>
<span class="icon-bar"></span>
</button>
<a class="navbar-brand" href="DimRed-bs.html">Data Analysis and Machine Learning: Preprocessing and Dimensionality Reduction</a>
</div>
<div class="navbar-collapse collapse navbar-responsive-collapse">
<ul class="nav navbar-nav navbar-right">
<li class="dropdown">
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
<ul class="dropdown-menu">
<!-- navigation toc: --> <li><a href="._DimRed-bs001.html#___sec0" style="font-size: 80%;"><b>Reducing the number of degrees of freedom, overarching view</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs002.html#___sec1" style="font-size: 80%;"><b>Preprocessing our data</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs003.html#___sec2" style="font-size: 80%;"><b>More preprocessing</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs004.html#___sec3" style="font-size: 80%;"><b>Simple preprocessing examples, Franke function and regression</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs005.html#___sec4" style="font-size: 80%;"><b>Simple preprocessing examples, breast cancer data and classification, Support Vector Machines</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs006.html#___sec5" style="font-size: 80%;"><b>More on Cancer Data, now with Logistic Regression</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs007.html#___sec6" style="font-size: 80%;"><b>Why should we think of reducing the dimensionality</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs008.html#___sec7" style="font-size: 80%;"><b>Basic ideas of the Principal Component Analysis (PCA)</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs009.html#___sec8" style="font-size: 80%;"><b>Introducing the Covariance and Correlation functions</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs010.html#___sec9" style="font-size: 80%;"><b>Correlation Function and Design/Feature Matrix</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs011.html#___sec10" style="font-size: 80%;"><b>Covariance Matrix Examples</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs012.html#___sec11" style="font-size: 80%;"><b>Correlation Matrix</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs013.html#___sec12" style="font-size: 80%;"><b>Correlation Matrix with Pandas</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs014.html#___sec13" style="font-size: 80%;"><b>Correlation Matrix with Pandas and the Franke function</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs015.html#___sec14" style="font-size: 80%;"><b>Rewriting the Covariance and/or Correlation Matrix</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs016.html#___sec15" style="font-size: 80%;"><b>Towards the PCA theorem</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs017.html#___sec16" style="font-size: 80%;"><b>The Algorithm before the Theorem</b></a></li>
<!-- navigation toc: --> <li><a href="#___sec17" style="font-size: 80%;"><b>Writing our own PCA code</b></a></li>
<!-- navigation toc: --> <li><a href="#___sec18" style="font-size: 80%;">&nbsp;&nbsp;&nbsp;Compute the sample mean and center the data</a></li>
<!-- navigation toc: --> <li><a href="#___sec19" style="font-size: 80%;">&nbsp;&nbsp;&nbsp;Compute the sample covariance</a></li>
<!-- navigation toc: --> <li><a href="#___sec20" style="font-size: 80%;">&nbsp;&nbsp;&nbsp;Diagonalize the sample covariance matrix to obtain the principal components</a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs019.html#___sec21" style="font-size: 80%;"><b>Classical PCA Theorem</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs020.html#___sec22" style="font-size: 80%;"><b>Proof of the PCA Theorem</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs021.html#___sec23" style="font-size: 80%;"><b>PCA Proof continued</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs022.html#___sec24" style="font-size: 80%;"><b>The final step</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs023.html#___sec25" style="font-size: 80%;"><b>Geometric Interpretation and link with Singular Value Decomposition</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs024.html#___sec26" style="font-size: 80%;"><b>Principal Component Analysis</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs025.html#___sec27" style="font-size: 80%;"><b>PCA and scikit-learn</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs026.html#___sec28" style="font-size: 80%;"><b>Back to the Cancer Data</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs027.html#___sec29" style="font-size: 80%;"><b>More on the PCA</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs028.html#___sec30" style="font-size: 80%;"><b>Incremental PCA</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs029.html#___sec31" style="font-size: 80%;"><b>Randomized PCA</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs030.html#___sec32" style="font-size: 80%;"><b>Kernel PCA</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs031.html#___sec33" style="font-size: 80%;"><b>LLE</b></a></li>
<!-- navigation toc: --> <li><a href="._DimRed-bs032.html#___sec34" style="font-size: 80%;"><b>Other techniques</b></a></li>
</ul>
</li>
</ul>
</div>
</div>
</div> <!-- end of navigation bar -->
<div class="container">
<p>&nbsp;</p><p>&nbsp;</p><p>&nbsp;</p> <!-- add vertical space -->
<a name="part0018"></a>
<!-- !split -->
<h2 id="___sec17" class="anchor">Writing our own PCA code </h2>
<p>
We will use a simple example first with two-dimensional data
drawn from a multivariate normal distribution with the following mean and covariance matrix:
$$
\mu = (-1,2) \qquad \Sigma = \begin{bmatrix} 4 & 2 \\
2 & 2
\end{bmatrix}
$$
Note that the mean refers to each column of data.
We will generate \( n = 1000 \) points \( X = \{ x_1, \ldots, x_N \} \) from
this distribution, and store them in the \( 1000 \times 2 \) matrix \( \boldsymbol{X} \).
<p>
The following Python code aids in setting up the data and writing out the design matrix.
Note that the function <b>multivariate</b> returns also the covariance discussed above and that it is defined by dividing by \( n-1 \) instead of \( n \).
<p>
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">IPython.display</span> <span style="color: #008000; font-weight: bold">import</span> display
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
mean <span style="color: #666666">=</span> (<span style="color: #666666">-1</span>, <span style="color: #666666">2</span>)
cov <span style="color: #666666">=</span> [[<span style="color: #666666">4</span>, <span style="color: #666666">2</span>], [<span style="color: #666666">2</span>, <span style="color: #666666">2</span>]]
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>multivariate_normal(mean, cov, n)
<span style="color: #408080; font-style: italic"># Print the X-matrix</span>
<span style="color: #008000; font-weight: bold">print</span>(X)
</pre></div>
<p>
Try to add to this code your own calculation of the covariance matrix.
<p>
Now we are going to implement the PCA algorithm. We will break it down into various substeps.
<h3 id="___sec18" class="anchor">Compute the sample mean and center the data </h3>
<p>
The first step of PCA is to compute the sample mean of the data and use it to center the data. Recall that the sample mean is
$$
\mu_n = \frac{1}{n} \sum_{i=1}^n x_i
$$
and the mean-centered data \( \bar{X} = \{ \bar{x}_1, \ldots, \bar{x}_n \} \) takes the form
$$
\bar{x}_i = x_i - \mu_n.
$$
When you are done with these steps, print out \( \mu_n \) to verify it is
close to \( \mu \) and plot your mean centered data to verify it is
centered at the origin! Compare your code with the functionality from <b>Scikit-Learn</b> discussed above.
The following code elements perform these operations using <b>pandas</b> or using our own functionality for doing so. The latter, using <b>numpy</b> is rather simply through the <b>mean()</b> function.
<p>
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>df <span style="color: #666666">=</span> pd<span style="color: #666666">.</span>DataFrame(X)
<span style="color: #408080; font-style: italic"># Pandas does the centering for us</span>
df <span style="color: #666666">=</span> df <span style="color: #666666">-</span>df<span style="color: #666666">.</span>mean()
display(df)
<span style="color: #408080; font-style: italic"># we center it ourselves</span>
X_centered <span style="color: #666666">=</span> X <span style="color: #666666">-</span> X<span style="color: #666666">.</span>mean(axis<span style="color: #666666">=0</span>)
<span style="color: #408080; font-style: italic"># test that we get the same as Pandas</span>
<span style="color: #008000; font-weight: bold">print</span>(X_centered<span style="color: #666666">-</span>df)
</pre></div>
<p>
Alternatively, you could also have used the functions we discussed earlier for scaling the data set.
That is, we could have used the <b>StandardScaler</b> function in <b>Scikit-Learn</b>, a function which
ensures that for each feature/predictor we study the mean value is
zero and the variance is one (every column in the design/feature
matrix). You would then not get the same results, since we divide by the variance. For diagonal covariance matrix elements will then be one, while the non-diagonal ones will be divided by \( 2\sqrt{2} \) for our specific case.
<h3 id="___sec19" class="anchor">Compute the sample covariance </h3>
<p>
Now we are going to use the mean centered data to compute the sample covariance of the data. Recall it is given by:
$$
\begin{equation*}
\Sigma_n = \frac{1}{n-1} \sum_{i=1}^n \bar{x}_i^T \bar{x}_i = \frac{1}{n-1} \sum_{i=1}^n (x_i - \mu_n)^T (x_i - \mu_n)
\end{equation*}
$$
where the data points \( x_i \in \mathbb{R}^p \) (here in this example \( p = 2 \)) are column vectors and \( x^T \) is the transpose of \( x \).
We can write our own code or simply use either the functionaly of <b>numpy</b> or that of <b>pandas</b>, as follows
<p>
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">print</span>(np<span style="color: #666666">.</span>cov(X<span style="color: #666666">.</span>T))
<span style="color: #008000; font-weight: bold">print</span>(df<span style="color: #666666">.</span>cov())
<span style="color: #008000; font-weight: bold">print</span>(np<span style="color: #666666">.</span>cov(X_centered<span style="color: #666666">.</span>T))
</pre></div>
<p>
Depending on the number of points \( n \), we will get results that are close to the covariance values defined above.
<h3 id="___sec20" class="anchor">Diagonalize the sample covariance matrix to obtain the principal components </h3>
<p>
Now we are ready to solve for the principal components! To do so we
diagonalize the sample covariance matrix \( \Sigma_n \). We can use the
function <b>np.linalg.eig</b> to do so. It will return the eigenvalues and
eigenvectors of \( \Sigma_n \). Once we have these we can perform the
following tasks:
<ul>
<li> We compute the percentage of the total variance captured by the first principal component</li>
<li> We plot the mean centered data and lines along the first and second principal components</li>
<li> Then we project the mean centered data onto the first and second principal components, and plot the projected data.</li>
<li> Finally, we approximate the data as</li>
</ul>
$$
\begin{equation*}
x_i \approx \tilde{x}_i = \mu_n + \langle x_i, v_0 \rangle v_0
\end{equation*}
$$
where \( v_0 \) is the first principal component.
<p>
Collecting all these steps we can write our own PCA function and
compare this with the functionality included in <b>Scikit-Learn</b>.
<p>
The code here outlines some of the elements we could include in the analysis. Feel free to extend upon this.
<p>
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #408080; font-style: italic">#Now we do an SVD</span>
U, s, V <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>svd(X_centered)
c1 <span style="color: #666666">=</span> V<span style="color: #666666">.</span>T[:, <span style="color: #666666">0</span>]
c2 <span style="color: #666666">=</span> V<span style="color: #666666">.</span>T[:, <span style="color: #666666">1</span>]
W2 <span style="color: #666666">=</span> V<span style="color: #666666">.</span>T[:, :<span style="color: #666666">2</span>]
X2D <span style="color: #666666">=</span> X_centered<span style="color: #666666">.</span>dot(W2)
<span style="color: #008000; font-weight: bold">print</span>(X2D)
<span style="color: #408080; font-style: italic">#thereafter we do a PCA with Scikit-learn</span>
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.decomposition</span> <span style="color: #008000; font-weight: bold">import</span> PCA
pca <span style="color: #666666">=</span> PCA(n_components <span style="color: #666666">=</span> <span style="color: #666666">2</span>)
X2Dsl <span style="color: #666666">=</span> pca<span style="color: #666666">.</span>fit_transform(X)
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">&quot;Check that we get the same&quot;</span>)
<span style="color: #008000; font-weight: bold">print</span>(X2D<span style="color: #666666">-</span>X2Dsl)
<span style="color: #008000; font-weight: bold">print</span>(pca<span style="color: #666666">.</span>components_<span style="color: #666666">.</span>T[:, <span style="color: #666666">0</span>])
</pre></div>
<p>
<p>
<!-- navigation buttons at the bottom of the page -->
<ul class="pagination">
<li><a href="._DimRed-bs017.html">&laquo;</a></li>
<li><a href="._DimRed-bs000.html">1</a></li>
<li><a href="">...</a></li>
<li><a href="._DimRed-bs010.html">11</a></li>
<li><a href="._DimRed-bs011.html">12</a></li>
<li><a href="._DimRed-bs012.html">13</a></li>
<li><a href="._DimRed-bs013.html">14</a></li>
<li><a href="._DimRed-bs014.html">15</a></li>
<li><a href="._DimRed-bs015.html">16</a></li>
<li><a href="._DimRed-bs016.html">17</a></li>
<li><a href="._DimRed-bs017.html">18</a></li>
<li class="active"><a href="._DimRed-bs018.html">19</a></li>
<li><a href="._DimRed-bs019.html">20</a></li>
<li><a href="._DimRed-bs020.html">21</a></li>
<li><a href="._DimRed-bs021.html">22</a></li>
<li><a href="._DimRed-bs022.html">23</a></li>
<li><a href="._DimRed-bs023.html">24</a></li>
<li><a href="._DimRed-bs024.html">25</a></li>
<li><a href="._DimRed-bs025.html">26</a></li>
<li><a href="._DimRed-bs026.html">27</a></li>
<li><a href="._DimRed-bs027.html">28</a></li>
<li><a href="">...</a></li>
<li><a href="._DimRed-bs032.html">33</a></li>
<li><a href="._DimRed-bs019.html">&raquo;</a></li>
</ul>
<!-- ------------------- end of main content --------------- -->
</div> <!-- end container -->
<!-- include javascript, jQuery *first* -->
<script src="https://ajax.googleapis.com/ajax/libs/jquery/1.10.2/jquery.min.js"></script>
<script src="https://netdna.bootstrapcdn.com/bootstrap/3.0.0/js/bootstrap.min.js"></script>
<!-- Bootstrap footer
<footer>
<a href="http://..."><img width="250" align=right src="http://..."></a>
</footer>
-->
<center style="font-size:80%">
<!-- copyright only on the titlepage -->
</center>
</body>
</html>