updating week 46

This commit is contained in:
mhjensen
2020-11-08 22:48:11 +01:00
parent 3e13bb0d64
commit 489c98a82f
34 changed files with 2834 additions and 2713 deletions
+121 -90
View File
@@ -7,9 +7,9 @@ Automatically generated HTML file from DocOnce source
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<meta name="description" content="Week 46: Support Vector Machines">
<meta name="description" content="Week 46: Gradient Boosting Summary and Support Vector Machines">
<title>Week 46: Support Vector Machines</title>
<title>Week 46: Gradient Boosting Summary and Support Vector Machines</title>
<!-- Bootstrap style: bootstrap -->
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
@@ -41,39 +41,40 @@ Automatically generated HTML file from DocOnce source
<!-- tocinfo
{'highest level': 2,
'sections': [('Support Vector Machines, overarching aims', 2, None, '___sec0'),
('Hyperplanes and all that', 2, None, '___sec1'),
('What is a hyperplane?', 2, None, '___sec2'),
('A $p$-dimensional space of features', 2, None, '___sec3'),
('The two-dimensional case', 2, None, '___sec4'),
('Getting into the details', 2, None, '___sec5'),
('First attempt at a minimization approach', 2, None, '___sec6'),
('Solving the equations', 2, None, '___sec7'),
('Code Example', 2, None, '___sec8'),
('Problems with the Simpler Approach', 2, None, '___sec9'),
('A better approach', 2, None, '___sec10'),
'sections': [('Overview of week 46', 2, None, '___sec0'),
('Support Vector Machines, overarching aims', 2, None, '___sec1'),
('Hyperplanes and all that', 2, None, '___sec2'),
('What is a hyperplane?', 2, None, '___sec3'),
('A $p$-dimensional space of features', 2, None, '___sec4'),
('The two-dimensional case', 2, None, '___sec5'),
('Getting into the details', 2, None, '___sec6'),
('First attempt at a minimization approach', 2, None, '___sec7'),
('Solving the equations', 2, None, '___sec8'),
('Code Example', 2, None, '___sec9'),
('Problems with the Simpler Approach', 2, None, '___sec10'),
('A better approach', 2, None, '___sec11'),
('A quick Reminder on Lagrangian Multipliers',
2,
None,
'___sec11'),
('Adding the Multiplier', 2, None, '___sec12'),
('Setting up the Problem', 2, None, '___sec13'),
('The problem to solve', 2, None, '___sec14'),
('The last steps', 2, None, '___sec15'),
('A soft classifier', 2, None, '___sec16'),
('Soft optmization problem', 2, None, '___sec17'),
('Kernels and non-linearity', 2, None, '___sec18'),
('The equations', 2, None, '___sec19'),
('The problem to solve', 2, None, '___sec20'),
("Different kernels and Mercer's theorem", 2, None, '___sec21'),
('The moons example', 2, None, '___sec22'),
'___sec12'),
('Adding the Multiplier', 2, None, '___sec13'),
('Setting up the Problem', 2, None, '___sec14'),
('The problem to solve', 2, None, '___sec15'),
('The last steps', 2, None, '___sec16'),
('A soft classifier', 2, None, '___sec17'),
('Soft optmization problem', 2, None, '___sec18'),
('Kernels and non-linearity', 2, None, '___sec19'),
('The equations', 2, None, '___sec20'),
('The problem to solve', 2, None, '___sec21'),
("Different kernels and Mercer's theorem", 2, None, '___sec22'),
('The moons example', 2, None, '___sec23'),
('Mathematical optimization of convex functions',
2,
None,
'___sec23'),
('How do we solve these problems?', 2, None, '___sec24'),
('A simple example', 2, None, '___sec25'),
('Back to the more realistic cases', 2, None, '___sec26')]}
'___sec24'),
('How do we solve these problems?', 2, None, '___sec25'),
('A simple example', 2, None, '___sec26'),
('Back to the more realistic cases', 2, None, '___sec27')]}
end of tocinfo -->
<body>
@@ -103,7 +104,7 @@ MathJax.Hub.Config({
<span class="icon-bar"></span>
<span class="icon-bar"></span>
</button>
<a class="navbar-brand" href="week46-bs.html">Week 46: Support Vector Machines</a>
<a class="navbar-brand" href="week46-bs.html">Week 46: Gradient Boosting Summary and Support Vector Machines</a>
</div>
<div class="navbar-collapse collapse navbar-responsive-collapse">
@@ -111,33 +112,34 @@ MathJax.Hub.Config({
<li class="dropdown">
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
<ul class="dropdown-menu">
<!-- navigation toc: --> <li><a href="._week46-bs001.html#___sec0" style="font-size: 80%;">Support Vector Machines, overarching aims</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs002.html#___sec1" style="font-size: 80%;">Hyperplanes and all that</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs003.html#___sec2" style="font-size: 80%;">What is a hyperplane?</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs004.html#___sec3" style="font-size: 80%;">A \( p \)-dimensional space of features</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs005.html#___sec4" style="font-size: 80%;">The two-dimensional case</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs006.html#___sec5" style="font-size: 80%;">Getting into the details</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs007.html#___sec6" style="font-size: 80%;">First attempt at a minimization approach</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs008.html#___sec7" style="font-size: 80%;">Solving the equations</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs009.html#___sec8" style="font-size: 80%;">Code Example</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs010.html#___sec9" style="font-size: 80%;">Problems with the Simpler Approach</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs011.html#___sec10" style="font-size: 80%;">A better approach</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs012.html#___sec11" style="font-size: 80%;">A quick Reminder on Lagrangian Multipliers</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs013.html#___sec12" style="font-size: 80%;">Adding the Multiplier</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs014.html#___sec13" style="font-size: 80%;">Setting up the Problem</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs015.html#___sec14" style="font-size: 80%;">The problem to solve</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs016.html#___sec15" style="font-size: 80%;">The last steps</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs017.html#___sec16" style="font-size: 80%;">A soft classifier</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs018.html#___sec17" style="font-size: 80%;">Soft optmization problem</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs019.html#___sec18" style="font-size: 80%;">Kernels and non-linearity</a></li>
<!-- navigation toc: --> <li><a href="#___sec19" style="font-size: 80%;">The equations</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs021.html#___sec20" style="font-size: 80%;">The problem to solve</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs022.html#___sec21" style="font-size: 80%;">Different kernels and Mercer's theorem</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs023.html#___sec22" style="font-size: 80%;">The moons example</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs024.html#___sec23" style="font-size: 80%;">Mathematical optimization of convex functions</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs025.html#___sec24" style="font-size: 80%;">How do we solve these problems?</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs026.html#___sec25" style="font-size: 80%;">A simple example</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs027.html#___sec26" style="font-size: 80%;">Back to the more realistic cases</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs001.html#___sec0" style="font-size: 80%;">Overview of week 46</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs002.html#___sec1" style="font-size: 80%;">Support Vector Machines, overarching aims</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs003.html#___sec2" style="font-size: 80%;">Hyperplanes and all that</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs004.html#___sec3" style="font-size: 80%;">What is a hyperplane?</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs005.html#___sec4" style="font-size: 80%;">A \( p \)-dimensional space of features</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs006.html#___sec5" style="font-size: 80%;">The two-dimensional case</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs007.html#___sec6" style="font-size: 80%;">Getting into the details</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs008.html#___sec7" style="font-size: 80%;">First attempt at a minimization approach</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs009.html#___sec8" style="font-size: 80%;">Solving the equations</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs010.html#___sec9" style="font-size: 80%;">Code Example</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs011.html#___sec10" style="font-size: 80%;">Problems with the Simpler Approach</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs012.html#___sec11" style="font-size: 80%;">A better approach</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs013.html#___sec12" style="font-size: 80%;">A quick Reminder on Lagrangian Multipliers</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs014.html#___sec13" style="font-size: 80%;">Adding the Multiplier</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs015.html#___sec14" style="font-size: 80%;">Setting up the Problem</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs016.html#___sec15" style="font-size: 80%;">The problem to solve</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs017.html#___sec16" style="font-size: 80%;">The last steps</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs018.html#___sec17" style="font-size: 80%;">A soft classifier</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs019.html#___sec18" style="font-size: 80%;">Soft optmization problem</a></li>
<!-- navigation toc: --> <li><a href="#___sec19" style="font-size: 80%;">Kernels and non-linearity</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs021.html#___sec20" style="font-size: 80%;">The equations</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs022.html#___sec21" style="font-size: 80%;">The problem to solve</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs023.html#___sec22" style="font-size: 80%;">Different kernels and Mercer's theorem</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs024.html#___sec23" style="font-size: 80%;">The moons example</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs025.html#___sec24" style="font-size: 80%;">Mathematical optimization of convex functions</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs026.html#___sec25" style="font-size: 80%;">How do we solve these problems?</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs027.html#___sec26" style="font-size: 80%;">A simple example</a></li>
<!-- navigation toc: --> <li><a href="._week46-bs028.html#___sec27" style="font-size: 80%;">Back to the more realistic cases</a></li>
</ul>
</li>
@@ -153,48 +155,76 @@ MathJax.Hub.Config({
<a name="part0020"></a>
<!-- !split -->
<h2 id="___sec19" class="anchor">The equations </h2>
<h2 id="___sec19" class="anchor">Kernels and non-linearity </h2>
<p>
Suppose we define a polynomial transformation of degree two only (we continue to live in a plane with \( x_i \) and \( y_i \) as variables)
$$
z = \phi(x_i) =\left(x_i^2, y_i^2, \sqrt{2}x_iy_i\right).
$$
The cases we have studied till now, were all characterized by two classes
with a close to linear separability. The classifiers we have described
so far find linear boundaries in our input feature space. It is
possible to make our procedure more flexible by exploring the feature
space using other basis expansions such as higher-order polynomials,
wavelets, splines etc.
<p>
With our new basis, the equations we solved earlier are basically the same, that is we have now (without the slack option for simplicity)
$$
{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{z}_i^T\boldsymbol{z}_j,
$$
subject to the constraints \( \lambda_i\geq 0 \), \( \sum_i\lambda_iy_i=0 \), and for the support vectors
$$
y_i(\boldsymbol{w}^T\boldsymbol{z}_i+b)= 1 \hspace{0.1cm}\forall i,
$$
from which we also find \( b \).
To compute \( \boldsymbol{z}_i^T\boldsymbol{z}_j \) we define the kernel \( K(\boldsymbol{x}_i,\boldsymbol{x}_j) \) as
$$
K(\boldsymbol{x}_i,\boldsymbol{x}_j)=\boldsymbol{z}_i^T\boldsymbol{z}_j= \phi(\boldsymbol{x}_i)^T\phi(\boldsymbol{x}_j).
$$
For the above example, the kernel reads
$$
K(\boldsymbol{x}_i,\boldsymbol{x}_j)=[x_i^2, y_i^2, \sqrt{2}x_iy_i]^T\begin{bmatrix} x_j^2 \\ y_j^2 \\ \sqrt{2}x_jy_j \end{bmatrix}=x_i^2x_j^2+2x_ix_jy_iy_j+y_i^2y_j^2.
$$
If our feature space is not easy to separate, as shown in the figure
here, we can achieve a better separation by introducing more complex
basis functions. The ideal would be, as shown in the next figure, to, via a specific transformation to
obtain a separation between the classes which is almost linear.
<p>
We note that this is nothing but the dot product of the two original
vectors \( (\boldsymbol{x}_i^T\boldsymbol{x}_j)^2 \). Instead of thus computing the
product in the Lagrangian of \( \boldsymbol{z}_i^T\boldsymbol{z}_j \) we simply compute
the dot product \( (\boldsymbol{x}_i^T\boldsymbol{x}_j)^2 \).
The change of basis, from \( x\rightarrow z=\phi(x) \) leads to the same type of equations to be solved, except that
we need to introduce for example a polynomial transformation to a two-dimensional training set.
<p>
This leads to the so-called
kernel trick and the result leads to the same as if we went through
the trouble of performing the transformation
\( \phi(\boldsymbol{x}_i)^T\phi(\boldsymbol{x}_j) \) during the SVM calculations.
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">os</span>
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">42</span>)
<span style="color: #408080; font-style: italic"># To plot pretty figures</span>
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib</span>
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
plt<span style="color: #666666">.</span>rcParams[<span style="color: #BA2121">&#39;axes.labelsize&#39;</span>] <span style="color: #666666">=</span> <span style="color: #666666">14</span>
plt<span style="color: #666666">.</span>rcParams[<span style="color: #BA2121">&#39;xtick.labelsize&#39;</span>] <span style="color: #666666">=</span> <span style="color: #666666">12</span>
plt<span style="color: #666666">.</span>rcParams[<span style="color: #BA2121">&#39;ytick.labelsize&#39;</span>] <span style="color: #666666">=</span> <span style="color: #666666">12</span>
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.svm</span> <span style="color: #008000; font-weight: bold">import</span> SVC
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> datasets
X1D <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">-4</span>, <span style="color: #666666">4</span>, <span style="color: #666666">9</span>)<span style="color: #666666">.</span>reshape(<span style="color: #666666">-1</span>, <span style="color: #666666">1</span>)
X2D <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[X1D, X1D<span style="color: #666666">**2</span>]
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">0</span>, <span style="color: #666666">0</span>, <span style="color: #666666">1</span>, <span style="color: #666666">1</span>, <span style="color: #666666">1</span>, <span style="color: #666666">1</span>, <span style="color: #666666">1</span>, <span style="color: #666666">0</span>, <span style="color: #666666">0</span>])
plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">11</span>, <span style="color: #666666">4</span>))
plt<span style="color: #666666">.</span>subplot(<span style="color: #666666">121</span>)
plt<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>, which<span style="color: #666666">=</span><span style="color: #BA2121">&#39;both&#39;</span>)
plt<span style="color: #666666">.</span>axhline(y<span style="color: #666666">=0</span>, color<span style="color: #666666">=</span><span style="color: #BA2121">&#39;k&#39;</span>)
plt<span style="color: #666666">.</span>plot(X1D[:, <span style="color: #666666">0</span>][y<span style="color: #666666">==0</span>], np<span style="color: #666666">.</span>zeros(<span style="color: #666666">4</span>), <span style="color: #BA2121">&quot;bs&quot;</span>)
plt<span style="color: #666666">.</span>plot(X1D[:, <span style="color: #666666">0</span>][y<span style="color: #666666">==1</span>], np<span style="color: #666666">.</span>zeros(<span style="color: #666666">5</span>), <span style="color: #BA2121">&quot;g^&quot;</span>)
plt<span style="color: #666666">.</span>gca()<span style="color: #666666">.</span>get_yaxis()<span style="color: #666666">.</span>set_ticks([])
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r&quot;$x_1$&quot;</span>, fontsize<span style="color: #666666">=20</span>)
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">-4.5</span>, <span style="color: #666666">4.5</span>, <span style="color: #666666">-0.2</span>, <span style="color: #666666">0.2</span>])
plt<span style="color: #666666">.</span>subplot(<span style="color: #666666">122</span>)
plt<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>, which<span style="color: #666666">=</span><span style="color: #BA2121">&#39;both&#39;</span>)
plt<span style="color: #666666">.</span>axhline(y<span style="color: #666666">=0</span>, color<span style="color: #666666">=</span><span style="color: #BA2121">&#39;k&#39;</span>)
plt<span style="color: #666666">.</span>axvline(x<span style="color: #666666">=0</span>, color<span style="color: #666666">=</span><span style="color: #BA2121">&#39;k&#39;</span>)
plt<span style="color: #666666">.</span>plot(X2D[:, <span style="color: #666666">0</span>][y<span style="color: #666666">==0</span>], X2D[:, <span style="color: #666666">1</span>][y<span style="color: #666666">==0</span>], <span style="color: #BA2121">&quot;bs&quot;</span>)
plt<span style="color: #666666">.</span>plot(X2D[:, <span style="color: #666666">0</span>][y<span style="color: #666666">==1</span>], X2D[:, <span style="color: #666666">1</span>][y<span style="color: #666666">==1</span>], <span style="color: #BA2121">&quot;g^&quot;</span>)
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r&quot;$x_1$&quot;</span>, fontsize<span style="color: #666666">=20</span>)
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r&quot;$x_2$&quot;</span>, fontsize<span style="color: #666666">=20</span>, rotation<span style="color: #666666">=0</span>)
plt<span style="color: #666666">.</span>gca()<span style="color: #666666">.</span>get_yaxis()<span style="color: #666666">.</span>set_ticks([<span style="color: #666666">0</span>, <span style="color: #666666">4</span>, <span style="color: #666666">8</span>, <span style="color: #666666">12</span>, <span style="color: #666666">16</span>])
plt<span style="color: #666666">.</span>plot([<span style="color: #666666">-4.5</span>, <span style="color: #666666">4.5</span>], [<span style="color: #666666">6.5</span>, <span style="color: #666666">6.5</span>], <span style="color: #BA2121">&quot;r--&quot;</span>, linewidth<span style="color: #666666">=3</span>)
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">-4.5</span>, <span style="color: #666666">4.5</span>, <span style="color: #666666">-1</span>, <span style="color: #666666">17</span>])
plt<span style="color: #666666">.</span>subplots_adjust(right<span style="color: #666666">=1</span>)
plt<span style="color: #666666">.</span>show()
</pre></div>
<p>
<p>
<!-- navigation buttons at the bottom of the page -->
@@ -218,6 +248,7 @@ the trouble of performing the transformation
<li><a href="._week46-bs025.html">26</a></li>
<li><a href="._week46-bs026.html">27</a></li>
<li><a href="._week46-bs027.html">28</a></li>
<li><a href="._week46-bs028.html">29</a></li>
<li><a href="._week46-bs021.html">&raquo;</a></li>
</ul>
<!-- ------------------- end of main content --------------- -->