updating week 46
This commit is contained in:
@@ -7,9 +7,9 @@ Automatically generated HTML file from DocOnce source
|
||||
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
||||
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<meta name="description" content="Week 46: Support Vector Machines">
|
||||
<meta name="description" content="Week 46: Gradient Boosting Summary and Support Vector Machines">
|
||||
|
||||
<title>Week 46: Support Vector Machines</title>
|
||||
<title>Week 46: Gradient Boosting Summary and Support Vector Machines</title>
|
||||
|
||||
<!-- Bootstrap style: bootstrap -->
|
||||
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
|
||||
@@ -41,39 +41,40 @@ Automatically generated HTML file from DocOnce source
|
||||
|
||||
<!-- tocinfo
|
||||
{'highest level': 2,
|
||||
'sections': [('Support Vector Machines, overarching aims', 2, None, '___sec0'),
|
||||
('Hyperplanes and all that', 2, None, '___sec1'),
|
||||
('What is a hyperplane?', 2, None, '___sec2'),
|
||||
('A $p$-dimensional space of features', 2, None, '___sec3'),
|
||||
('The two-dimensional case', 2, None, '___sec4'),
|
||||
('Getting into the details', 2, None, '___sec5'),
|
||||
('First attempt at a minimization approach', 2, None, '___sec6'),
|
||||
('Solving the equations', 2, None, '___sec7'),
|
||||
('Code Example', 2, None, '___sec8'),
|
||||
('Problems with the Simpler Approach', 2, None, '___sec9'),
|
||||
('A better approach', 2, None, '___sec10'),
|
||||
'sections': [('Overview of week 46', 2, None, '___sec0'),
|
||||
('Support Vector Machines, overarching aims', 2, None, '___sec1'),
|
||||
('Hyperplanes and all that', 2, None, '___sec2'),
|
||||
('What is a hyperplane?', 2, None, '___sec3'),
|
||||
('A $p$-dimensional space of features', 2, None, '___sec4'),
|
||||
('The two-dimensional case', 2, None, '___sec5'),
|
||||
('Getting into the details', 2, None, '___sec6'),
|
||||
('First attempt at a minimization approach', 2, None, '___sec7'),
|
||||
('Solving the equations', 2, None, '___sec8'),
|
||||
('Code Example', 2, None, '___sec9'),
|
||||
('Problems with the Simpler Approach', 2, None, '___sec10'),
|
||||
('A better approach', 2, None, '___sec11'),
|
||||
('A quick Reminder on Lagrangian Multipliers',
|
||||
2,
|
||||
None,
|
||||
'___sec11'),
|
||||
('Adding the Multiplier', 2, None, '___sec12'),
|
||||
('Setting up the Problem', 2, None, '___sec13'),
|
||||
('The problem to solve', 2, None, '___sec14'),
|
||||
('The last steps', 2, None, '___sec15'),
|
||||
('A soft classifier', 2, None, '___sec16'),
|
||||
('Soft optmization problem', 2, None, '___sec17'),
|
||||
('Kernels and non-linearity', 2, None, '___sec18'),
|
||||
('The equations', 2, None, '___sec19'),
|
||||
('The problem to solve', 2, None, '___sec20'),
|
||||
("Different kernels and Mercer's theorem", 2, None, '___sec21'),
|
||||
('The moons example', 2, None, '___sec22'),
|
||||
'___sec12'),
|
||||
('Adding the Multiplier', 2, None, '___sec13'),
|
||||
('Setting up the Problem', 2, None, '___sec14'),
|
||||
('The problem to solve', 2, None, '___sec15'),
|
||||
('The last steps', 2, None, '___sec16'),
|
||||
('A soft classifier', 2, None, '___sec17'),
|
||||
('Soft optmization problem', 2, None, '___sec18'),
|
||||
('Kernels and non-linearity', 2, None, '___sec19'),
|
||||
('The equations', 2, None, '___sec20'),
|
||||
('The problem to solve', 2, None, '___sec21'),
|
||||
("Different kernels and Mercer's theorem", 2, None, '___sec22'),
|
||||
('The moons example', 2, None, '___sec23'),
|
||||
('Mathematical optimization of convex functions',
|
||||
2,
|
||||
None,
|
||||
'___sec23'),
|
||||
('How do we solve these problems?', 2, None, '___sec24'),
|
||||
('A simple example', 2, None, '___sec25'),
|
||||
('Back to the more realistic cases', 2, None, '___sec26')]}
|
||||
'___sec24'),
|
||||
('How do we solve these problems?', 2, None, '___sec25'),
|
||||
('A simple example', 2, None, '___sec26'),
|
||||
('Back to the more realistic cases', 2, None, '___sec27')]}
|
||||
end of tocinfo -->
|
||||
|
||||
<body>
|
||||
@@ -103,7 +104,7 @@ MathJax.Hub.Config({
|
||||
<span class="icon-bar"></span>
|
||||
<span class="icon-bar"></span>
|
||||
</button>
|
||||
<a class="navbar-brand" href="week46-bs.html">Week 46: Support Vector Machines</a>
|
||||
<a class="navbar-brand" href="week46-bs.html">Week 46: Gradient Boosting Summary and Support Vector Machines</a>
|
||||
</div>
|
||||
|
||||
<div class="navbar-collapse collapse navbar-responsive-collapse">
|
||||
@@ -111,33 +112,34 @@ MathJax.Hub.Config({
|
||||
<li class="dropdown">
|
||||
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
|
||||
<ul class="dropdown-menu">
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs001.html#___sec0" style="font-size: 80%;">Support Vector Machines, overarching aims</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs002.html#___sec1" style="font-size: 80%;">Hyperplanes and all that</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs003.html#___sec2" style="font-size: 80%;">What is a hyperplane?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs004.html#___sec3" style="font-size: 80%;">A \( p \)-dimensional space of features</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs005.html#___sec4" style="font-size: 80%;">The two-dimensional case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs006.html#___sec5" style="font-size: 80%;">Getting into the details</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs007.html#___sec6" style="font-size: 80%;">First attempt at a minimization approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs008.html#___sec7" style="font-size: 80%;">Solving the equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs009.html#___sec8" style="font-size: 80%;">Code Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs010.html#___sec9" style="font-size: 80%;">Problems with the Simpler Approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs011.html#___sec10" style="font-size: 80%;">A better approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs012.html#___sec11" style="font-size: 80%;">A quick Reminder on Lagrangian Multipliers</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs013.html#___sec12" style="font-size: 80%;">Adding the Multiplier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs014.html#___sec13" style="font-size: 80%;">Setting up the Problem</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs015.html#___sec14" style="font-size: 80%;">The problem to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#___sec15" style="font-size: 80%;">The last steps</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs017.html#___sec16" style="font-size: 80%;">A soft classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs018.html#___sec17" style="font-size: 80%;">Soft optmization problem</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs019.html#___sec18" style="font-size: 80%;">Kernels and non-linearity</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs020.html#___sec19" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs021.html#___sec20" style="font-size: 80%;">The problem to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs022.html#___sec21" style="font-size: 80%;">Different kernels and Mercer's theorem</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs023.html#___sec22" style="font-size: 80%;">The moons example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs024.html#___sec23" style="font-size: 80%;">Mathematical optimization of convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs025.html#___sec24" style="font-size: 80%;">How do we solve these problems?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs026.html#___sec25" style="font-size: 80%;">A simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs027.html#___sec26" style="font-size: 80%;">Back to the more realistic cases</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs001.html#___sec0" style="font-size: 80%;">Overview of week 46</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs002.html#___sec1" style="font-size: 80%;">Support Vector Machines, overarching aims</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs003.html#___sec2" style="font-size: 80%;">Hyperplanes and all that</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs004.html#___sec3" style="font-size: 80%;">What is a hyperplane?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs005.html#___sec4" style="font-size: 80%;">A \( p \)-dimensional space of features</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs006.html#___sec5" style="font-size: 80%;">The two-dimensional case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs007.html#___sec6" style="font-size: 80%;">Getting into the details</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs008.html#___sec7" style="font-size: 80%;">First attempt at a minimization approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs009.html#___sec8" style="font-size: 80%;">Solving the equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs010.html#___sec9" style="font-size: 80%;">Code Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs011.html#___sec10" style="font-size: 80%;">Problems with the Simpler Approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs012.html#___sec11" style="font-size: 80%;">A better approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs013.html#___sec12" style="font-size: 80%;">A quick Reminder on Lagrangian Multipliers</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs014.html#___sec13" style="font-size: 80%;">Adding the Multiplier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs015.html#___sec14" style="font-size: 80%;">Setting up the Problem</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#___sec15" style="font-size: 80%;">The problem to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs017.html#___sec16" style="font-size: 80%;">The last steps</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs018.html#___sec17" style="font-size: 80%;">A soft classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs019.html#___sec18" style="font-size: 80%;">Soft optmization problem</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs020.html#___sec19" style="font-size: 80%;">Kernels and non-linearity</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs021.html#___sec20" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs022.html#___sec21" style="font-size: 80%;">The problem to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs023.html#___sec22" style="font-size: 80%;">Different kernels and Mercer's theorem</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs024.html#___sec23" style="font-size: 80%;">The moons example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs025.html#___sec24" style="font-size: 80%;">Mathematical optimization of convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs026.html#___sec25" style="font-size: 80%;">How do we solve these problems?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs027.html#___sec26" style="font-size: 80%;">A simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week46-bs028.html#___sec27" style="font-size: 80%;">Back to the more realistic cases</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -153,36 +155,26 @@ MathJax.Hub.Config({
|
||||
<a name="part0016"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="___sec15" class="anchor">The last steps </h2>
|
||||
<h2 id="___sec15" class="anchor">The problem to solve </h2>
|
||||
|
||||
<p>
|
||||
Solving the above problem, yields the values of \( \lambda_i \).
|
||||
To find the coefficients of your hyperplane we need simply to compute
|
||||
We can rewrite
|
||||
$$
|
||||
\boldsymbol{w}=\sum_{i} \lambda_iy_i\boldsymbol{x}_i.
|
||||
{\cal L}=\sum_i\lambda_i-\frac{1}{2}\sum_{ij}^n\lambda_i\lambda_jy_iy_j\boldsymbol{x}_i^T\boldsymbol{x}_j,
|
||||
$$
|
||||
|
||||
With our vector \( \boldsymbol{w} \) we can in turn find the value of the intercept \( b \) (here in two dimensions) via
|
||||
and its constraints in terms of a matrix-vector problem where we minimize w.r.t. \( \lambda \) the following problem
|
||||
$$
|
||||
y_i(\boldsymbol{w}^T\boldsymbol{x}_i+b)=1,
|
||||
\frac{1}{2} \boldsymbol{\lambda}^T\begin{bmatrix} y_1y_1\boldsymbol{x}_1^T\boldsymbol{x}_1 & y_1y_2\boldsymbol{x}_1^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_1^T\boldsymbol{x}_n \\
|
||||
y_2y_1\boldsymbol{x}_2^T\boldsymbol{x}_1 & y_2y_2\boldsymbol{x}_2^T\boldsymbol{x}_2 & \dots & \dots & y_1y_n\boldsymbol{x}_2^T\boldsymbol{x}_n \\
|
||||
\dots & \dots & \dots & \dots & \dots \\
|
||||
\dots & \dots & \dots & \dots & \dots \\
|
||||
y_ny_1\boldsymbol{x}_n^T\boldsymbol{x}_1 & y_ny_2\boldsymbol{x}_n^T\boldsymbol{x}_2 & \dots & \dots & y_ny_n\boldsymbol{x}_n^T\boldsymbol{x}_n \\
|
||||
\end{bmatrix}\boldsymbol{\lambda}-\mathbb{1}\boldsymbol{\lambda},
|
||||
$$
|
||||
|
||||
resulting in
|
||||
$$
|
||||
b = \frac{1}{y_i}-\boldsymbol{w}^T\boldsymbol{x}_i,
|
||||
$$
|
||||
|
||||
or if we write it out in terms of the support vectors only, with \( N_s \) being their number, we have
|
||||
$$
|
||||
b = \frac{1}{N_s}\sum_{j\in N_s}\left(y_j-\sum_{i=1}^n\lambda_iy_i\boldsymbol{x}_i^T\boldsymbol{x}_j\right).
|
||||
$$
|
||||
|
||||
With our hyperplane coefficients we can use our classifier to assign any observation by simply using
|
||||
$$
|
||||
y_i = \mathrm{sign}(\boldsymbol{w}^T\boldsymbol{x}_i+b).
|
||||
$$
|
||||
|
||||
Below we discuss how to find the optimal values of \( \lambda_i \). Before we proceed however, we discuss now the so-called soft classifier.
|
||||
subject to \( \boldsymbol{y}^T\boldsymbol{\lambda}=0 \). Here we defined the vectors \( \boldsymbol{\lambda} =[\lambda_1,\lambda_2,\dots,\lambda_n] \) and
|
||||
\( \boldsymbol{y}=[y_1,y_2,\dots,y_n] \).
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -210,7 +202,7 @@ Below we discuss how to find the optimal values of \( \lambda_i \). Before we pr
|
||||
<li><a href="._week46-bs024.html">25</a></li>
|
||||
<li><a href="._week46-bs025.html">26</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week46-bs027.html">28</a></li>
|
||||
<li><a href="._week46-bs028.html">29</a></li>
|
||||
<li><a href="._week46-bs017.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
Reference in New Issue
Block a user