2383 lines
101 KiB
HTML
2383 lines
101 KiB
HTML
\
|
|
<!DOCTYPE html>
|
|
|
|
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
|
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
|
|
<meta name="description" content="Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods">
|
|
|
|
<title>Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods</title>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
<!-- reveal.js: http://lab.hakim.se/reveal-js/ -->
|
|
|
|
<meta name="viewport" content="width=device-width, initial-scale=1.0, maximum-scale=1.0, user-scalable=no">
|
|
|
|
<meta name="apple-mobile-web-app-capable" content="yes" />
|
|
<meta name="apple-mobile-web-app-status-bar-style" content="black-translucent" />
|
|
<meta name="viewport" content="width=device-width, initial-scale=1.0, maximum-scale=1.0, user-scalable=no, minimal-ui">
|
|
|
|
<link rel="stylesheet" href="reveal.js/css/reveal.css">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/beige.css" id="theme">
|
|
<!--
|
|
<link rel="stylesheet" href="reveal.js/css/reveal.css">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/beige.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/beigesmall.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/solarized.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/serif.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/night.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/moon.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/simple.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/sky.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/darkgray.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/default.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/cbc.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/simula.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/black.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/white.css" id="theme">
|
|
<link rel="stylesheet" href="reveal.js/css/theme/league.css" id="theme">
|
|
-->
|
|
|
|
<!-- For syntax highlighting -->
|
|
<link rel="stylesheet" href="reveal.js/lib/css/zenburn.css">
|
|
|
|
<!-- Printing and PDF exports -->
|
|
<script>
|
|
var link = document.createElement( 'link' );
|
|
link.rel = 'stylesheet';
|
|
link.type = 'text/css';
|
|
link.href = window.location.search.match( /print-pdf/gi ) ? 'css/print/pdf.css' : 'css/print/paper.css';
|
|
document.getElementsByTagName( 'head' )[0].appendChild( link );
|
|
</script>
|
|
|
|
<style type="text/css">
|
|
hr { border: 0; width: 80%; border-bottom: 1px solid #aaa}
|
|
p.caption { width: 80%; font-size: 60%; font-style: italic; text-align: left; }
|
|
hr.figure { border: 0; width: 80%; border-bottom: 1px solid #aaa}
|
|
.reveal .alert-text-small { font-size: 80%; }
|
|
.reveal .alert-text-large { font-size: 130%; }
|
|
.reveal .alert-text-normal { font-size: 90%; }
|
|
.reveal .alert {
|
|
padding:8px 35px 8px 14px; margin-bottom:18px;
|
|
text-shadow:0 1px 0 rgba(255,255,255,0.5);
|
|
border:5px solid #bababa;
|
|
-webkit-border-radius: 14px; -moz-border-radius: 14px;
|
|
border-radius:14px;
|
|
background-position: 10px 10px;
|
|
background-repeat: no-repeat;
|
|
background-size: 38px;
|
|
padding-left: 30px; /* 55px; if icon */
|
|
}
|
|
.reveal .alert-block {padding-top:14px; padding-bottom:14px}
|
|
.reveal .alert-block > p, .alert-block > ul {margin-bottom:1em}
|
|
/*.reveal .alert li {margin-top: 1em}*/
|
|
.reveal .alert-block p+p {margin-top:5px}
|
|
/*.reveal .alert-notice { background-image: url(http://hplgit.github.io/doconce/bundled/html_images/small_gray_notice.png); }
|
|
.reveal .alert-summary { background-image:url(http://hplgit.github.io/doconce/bundled/html_images/small_gray_summary.png); }
|
|
.reveal .alert-warning { background-image: url(http://hplgit.github.io/doconce/bundled/html_images/small_gray_warning.png); }
|
|
.reveal .alert-question {background-image:url(http://hplgit.github.io/doconce/bundled/html_images/small_gray_question.png); } */
|
|
|
|
</style>
|
|
|
|
|
|
|
|
<!-- Styles for table layout of slides -->
|
|
<style type="text/css">
|
|
td.padding {
|
|
padding-top:20px;
|
|
padding-bottom:20px;
|
|
padding-right:50px;
|
|
padding-left:50px;
|
|
}
|
|
</style>
|
|
|
|
</head>
|
|
|
|
<body>
|
|
<div class="reveal">
|
|
|
|
<!-- Any section element inside the <div class="slides"> container
|
|
is displayed as a slide -->
|
|
|
|
<div class="slides">
|
|
|
|
|
|
|
|
|
|
|
|
<script type="text/x-mathjax-config">
|
|
MathJax.Hub.Config({
|
|
TeX: {
|
|
equationNumbers: { autoNumber: "none" },
|
|
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
|
|
}
|
|
});
|
|
</script>
|
|
<script type="text/javascript" async
|
|
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
|
|
</script>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
<section>
|
|
<!-- ------------------- main content ---------------------- -->
|
|
|
|
|
|
|
|
<center><h1 style="text-align: center;">Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods</h1></center> <!-- document title -->
|
|
|
|
<p>
|
|
<!-- author(s): Morten Hjorth-Jensen -->
|
|
|
|
<center>
|
|
<b>Morten Hjorth-Jensen</b> [1, 2]
|
|
</center>
|
|
|
|
<p> <br>
|
|
<!-- institution(s) -->
|
|
|
|
<center>[1] <b>Department of Physics, University of Oslo</b></center>
|
|
<center>[2] <b>Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University</b></center>
|
|
<br>
|
|
<p> <br>
|
|
<center><h4>Oct 6, 2018</h4></center> <!-- date -->
|
|
<br>
|
|
<p>
|
|
|
|
<center style="font-size:80%">
|
|
<!-- copyright --> © 1999-2018, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
|
|
</center>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec0">Optimization, the central part of any Machine Learning algortithm </h2>
|
|
|
|
<p>
|
|
Almost every problem in machine learning and data science starts with
|
|
a dataset \( X \), a model \( g(\beta) \), which is a function of the
|
|
parameters \( \beta \) and a cost function \( C(X, g(\beta)) \) that allows
|
|
us to judge how well the model \( g(\beta) \) explains the observations
|
|
\( X \). The model is fit by finding the values of \( \beta \) that minimize
|
|
the cost function. Ideally we would be able to solve for \( \beta \)
|
|
analytically, however this is not possible in general and we must use
|
|
some approximative/numerical method to compute the minimum.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec1">Revisiting our Logistic Regression case </h2>
|
|
|
|
<p>
|
|
In our discussion on Logistic Regression we studied the
|
|
case of
|
|
two classes, with \( y_i \) either
|
|
\( 0 \) or \( 1 \). Furthermore we assumed also that we have only two
|
|
parameters \( \beta \) in our fitting, that is we
|
|
defined probabilities
|
|
|
|
<p> <br>
|
|
$$
|
|
\begin{align*}
|
|
p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\
|
|
p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}),
|
|
\end{align*}
|
|
$$
|
|
<p> <br>
|
|
|
|
where \( \hat{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec2">The equations to solve </h2>
|
|
|
|
<p>
|
|
Our compact equations used a definition of a vector \( \hat{y} \) with \( n \)
|
|
elements \( y_i \), an \( n\times p \) matrix \( \hat{X} \) which contains the
|
|
\( x_i \) values and a vector \( \hat{p} \) of fitted probabilities
|
|
\( p(y_i\vert x_i,\hat{\beta}) \). We rewrote in a more compact form
|
|
the first derivative of the cost function as
|
|
|
|
<p> <br>
|
|
$$
|
|
\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right).
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
If we in addition define a diagonal matrix \( \hat{W} \) with elements
|
|
\( p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta}) \), we can obtain a compact expression of the second derivative as
|
|
|
|
<p> <br>
|
|
$$
|
|
\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}.
|
|
$$
|
|
<p> <br>
|
|
|
|
This defines what is called the Hessian matrix.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec3">Solving using Newton-Raphson's method </h2>
|
|
|
|
<p>
|
|
If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives.
|
|
|
|
<p>
|
|
Our iterative scheme is then given by
|
|
|
|
<p> <br>
|
|
$$
|
|
\hat{\beta}^{\mathrm{new}} = \hat{\beta}^{\mathrm{old}}-\left(\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T}\right)^{-1}_{\hat{\beta}^{\mathrm{old}}}\times \left(\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}}\right)_{\hat{\beta}^{\mathrm{old}}},
|
|
$$
|
|
<p> <br>
|
|
|
|
or in matrix form as
|
|
|
|
<p> <br>
|
|
$$
|
|
\hat{\beta}^{\mathrm{new}} = \hat{\beta}^{\mathrm{old}}-\left(\hat{X}^T\hat{W}\hat{X} \right)^{-1}\times \left(-\hat{X}^T(\hat{y}-\hat{p}) \right)_{\hat{\beta}^{\mathrm{old}}}.
|
|
$$
|
|
<p> <br>
|
|
|
|
The right-hand side is computed with the old values of \( \beta \).
|
|
|
|
<p>
|
|
If we can compute these matrices, in particular the Hessian, the above is often the easiest method to implement.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec4">Brief reminder on Newton-Raphson's method </h2>
|
|
|
|
<p>
|
|
Let us quickly remind ourselves how we derive the above method.
|
|
|
|
<p>
|
|
Perhaps the most celebrated of all one-dimensional root-finding
|
|
routines is Newton's method, also called the Newton-Raphson
|
|
method. This method requires the evaluation of both the
|
|
function \( f \) and its derivative \( f' \) at arbitrary points.
|
|
If you can only calculate the derivative
|
|
numerically and/or your function is not of the smooth type, we
|
|
normally discourage the use of this method.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec5">The equations </h2>
|
|
|
|
<p>
|
|
The Newton-Raphson formula consists geometrically of extending the
|
|
tangent line at a current point until it crosses zero, then setting
|
|
the next guess to the abscissa of that zero-crossing. The mathematics
|
|
behind this method is rather simple. Employing a Taylor expansion for
|
|
\( x \) sufficiently close to the solution \( s \), we have
|
|
|
|
<p> <br>
|
|
$$
|
|
f(s)=0=f(x)+(s-x)f'(x)+\frac{(s-x)^2}{2}f''(x) +\dots.
|
|
\tag{1}
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
For small enough values of the function and for well-behaved
|
|
functions, the terms beyond linear are unimportant, hence we obtain
|
|
|
|
<p> <br>
|
|
$$
|
|
f(x)+(s-x)f'(x)\approx 0,
|
|
$$
|
|
<p> <br>
|
|
|
|
yielding
|
|
<p> <br>
|
|
$$
|
|
s\approx x-\frac{f(x)}{f'(x)}.
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
Having in mind an iterative procedure, it is natural to start iterating with
|
|
<p> <br>
|
|
$$
|
|
x_{n+1}=x_n-\frac{f(x_n)}{f'(x_n)}.
|
|
$$
|
|
<p> <br>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec6">Simple geometric interpretation </h2>
|
|
|
|
<p>
|
|
The above is Newton-Raphson's method. It has a simple geometric
|
|
interpretation, namely \( x_{n+1} \) is the point where the tangent from
|
|
\( (x_n,f(x_n)) \) crosses the \( x \)-axis. Close to the solution,
|
|
Newton-Raphson converges fast to the desired result. However, if we
|
|
are far from a root, where the higher-order terms in the series are
|
|
important, the Newton-Raphson formula can give grossly inaccurate
|
|
results. For instance, the initial guess for the root might be so far
|
|
from the true root as to let the search interval include a local
|
|
maximum or minimum of the function. If an iteration places a trial
|
|
guess near such a local extremum, so that the first derivative nearly
|
|
vanishes, then Newton-Raphson may fail totally
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec7">Extending to more than one variable </h2>
|
|
|
|
<p>
|
|
Newton's method can be generalized to systems of several non-linear equations
|
|
and variables. Consider the case with two equations
|
|
<p> <br>
|
|
$$
|
|
\begin{array}{cc} f_1(x_1,x_2) &=0\\
|
|
f_2(x_1,x_2) &=0,\end{array}
|
|
$$
|
|
<p> <br>
|
|
|
|
which we Taylor expand to obtain
|
|
|
|
<p> <br>
|
|
$$
|
|
\begin{array}{cc} 0=f_1(x_1+h_1,x_2+h_2)=&f_1(x_1,x_2)+h_1
|
|
\partial f_1/\partial x_1+h_2
|
|
\partial f_1/\partial x_2+\dots\\
|
|
0=f_2(x_1+h_1,x_2+h_2)=&f_2(x_1,x_2)+h_1
|
|
\partial f_2/\partial x_1+h_2
|
|
\partial f_2/\partial x_2+\dots
|
|
\end{array}.
|
|
$$
|
|
<p> <br>
|
|
|
|
Defining the Jacobian matrix \( {\bf \hat{J}} \) we have
|
|
<p> <br>
|
|
$$
|
|
{\bf \hat{J}}=\left( \begin{array}{cc}
|
|
\partial f_1/\partial x_1 & \partial f_1/\partial x_2 \\
|
|
\partial f_2/\partial x_1 &\partial f_2/\partial x_2
|
|
\end{array} \right),
|
|
$$
|
|
<p> <br>
|
|
|
|
we can rephrase Newton's method as
|
|
<p> <br>
|
|
$$
|
|
\left(\begin{array}{c} x_1^{n+1} \\ x_2^{n+1} \end{array} \right)=
|
|
\left(\begin{array}{c} x_1^{n} \\ x_2^{n} \end{array} \right)+
|
|
\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right),
|
|
$$
|
|
<p> <br>
|
|
|
|
where we have defined
|
|
<p> <br>
|
|
$$
|
|
\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right)=
|
|
-{\bf \hat{J}}^{-1}
|
|
\left(\begin{array}{c} f_1(x_1^{n},x_2^{n}) \\ f_2(x_1^{n},x_2^{n}) \end{array} \right).
|
|
$$
|
|
<p> <br>
|
|
|
|
We need thus to compute the inverse of the Jacobian matrix and it
|
|
is to understand that difficulties may
|
|
arise in case \( {\bf \hat{J}} \) is nearly singular.
|
|
|
|
<p>
|
|
It is rather straightforward to extend the above scheme to systems of
|
|
more than two non-linear equations. In our case, the Jacobian matrix is given by the Hessian that represents the second derivative of cost function.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec8">Steepest descent </h2>
|
|
|
|
<p>
|
|
The basic idea of gradient descent is
|
|
that a function \( F(\mathbf{x}) \),
|
|
\( \mathbf{x} \equiv (x_1,\cdots,x_n) \), decreases fastest if one goes from \( \bf {x} \) in the
|
|
direction of the negative gradient \( -\nabla F(\mathbf{x}) \).
|
|
|
|
<p>
|
|
It can be shown that if
|
|
<p> <br>
|
|
$$
|
|
\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k),
|
|
$$
|
|
<p> <br>
|
|
|
|
with \( \gamma_k > 0 \).
|
|
|
|
<p>
|
|
For \( \gamma_k \) small enough, then \( F(\mathbf{x}_{k+1}) \leq
|
|
F(\mathbf{x}_k) \). This means that for a sufficiently small \( \gamma_k \)
|
|
we are always moving towards smaller function values, i.e a minimum.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec9">More on Steepest descent </h2>
|
|
|
|
<p>
|
|
The previous observation is the basis of the method of steepest
|
|
descent, which is also referred to as just gradient descent (GD). One
|
|
starts with an initial guess \( \mathbf{x}_0 \) for a minimum of \( F \) and
|
|
computes new approximations according to
|
|
|
|
<p> <br>
|
|
$$
|
|
\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), \ \ k \geq 0.
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
The parameter \( \gamma_k \) is often referred to as the step length or
|
|
the learning rate within the context of Machine Learning.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec10">The ideal </h2>
|
|
|
|
<p>
|
|
Ideally the sequence \( \{\mathbf{x}_k \}_{k=0} \) converges to a global
|
|
minimum of the function \( F \). In general we do not know if we are in a
|
|
global or local minimum. In the special case when \( F \) is a convex
|
|
function, all local minima are also global minima, so in this case
|
|
gradient descent can converge to the global solution. The advantage of
|
|
this scheme is that it is conceptually simple and straightforward to
|
|
implement. However the method in this form has some severe
|
|
limitations:
|
|
|
|
<p>
|
|
In machine learing we are often faced with non-convex high dimensional
|
|
cost functions with many local minima. Since GD is deterministic we
|
|
will get stuck in a local minimum, if the method converges, unless we
|
|
have a very good intial guess. This also implies that the scheme is
|
|
sensitive to the chosen initial condition.
|
|
|
|
<p>
|
|
Note that the gradient is a function of \( \mathbf{x} =
|
|
(x_1,\cdots,x_n) \) which makes it expensive to compute numerically.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec11">The sensitiveness of the gradient descent </h2>
|
|
|
|
<p>
|
|
The gradient descent method
|
|
is sensitive to the choice of learning rate \( \gamma_k \). This is due
|
|
to the fact that we are only guaranteed that \( F(\mathbf{x}_{k+1}) \leq
|
|
F(\mathbf{x}_k) \) for sufficiently small \( \gamma_k \). The problem is to
|
|
determine an optimal learning rate. If the learning rate is chosen too
|
|
small the method will take a long time to converge and if it is too
|
|
large we can experience erratic behavior.
|
|
|
|
<p>
|
|
Many of these shortcomings can be alleviated by introducing
|
|
randomness. One such method is that of Stochastic Gradient Descent
|
|
(SGD), see below.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec12">Convex functions </h2>
|
|
|
|
<p>
|
|
Ideally we want our cost/loss function to be convex(concave).
|
|
|
|
<p>
|
|
First we give the definition of a convex set: A set \( C \) in
|
|
\( \mathbb{R}^n \) is said to be convex if, for all \( x \) and \( y \) in \( C \) and
|
|
all \( t \in (0,1) \) , the point \( (1 − t)x + ty \) also belongs to
|
|
C. Geometrically this means that every point on the line segment
|
|
connecting \( x \) and \( y \) is in \( C \) as discussed below.
|
|
|
|
<p>
|
|
The convex subsets of \( \mathbb{R} \) are the intervals of
|
|
\( \mathbb{R} \). Examples of convex sets of \( \mathbb{R}^2 \) are the
|
|
regular polygons (triangles, rectangles, pentagons, etc...).
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec13">Convex function </h2>
|
|
|
|
<p>
|
|
<b>Convex function</b>: Let \( X \subset \mathbb{R}^n \) be a convex set. Assume that the function \( f: X \rightarrow \mathbb{R} \) is continuous, then \( f \) is said to be convex if <p> <br>
|
|
$$f(tx_1 + (1-t)x_2) \leq tf(x_1) + (1-t)f(x_2) $$
|
|
<p> <br> for all \( x_1, x_2 \in X \) and for all \( t \in [0,1] \). If \( \leq \) is replaced with a strict inequaltiy in the definition, we demand \( x_1 \neq x_2 \) and \( t\in(0,1) \) then \( f \) is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting \( f(x_1) \) and \( f(x_2) \), the value of the function on the interval \( [x_1,x_2] \) is always below the line as illustrated below.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec14">Conditions on convex functions </h2>
|
|
|
|
<p>
|
|
In the following we state first and second-order conditions which
|
|
ensures convexity of a function \( f \). We write \( D_f \) to denote the
|
|
domain of \( f \), i.e the subset of \( R^n \) where \( f \) is defined. For more
|
|
details and proofs we refer to: <a href="http://stanford.edu/boyd/cvxbook/, 2004" target="_blank">S. Boyd and L. Vandenberghe. Convex Optimization. Cambridge University Press</a>.
|
|
|
|
<p>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b>First order condition.</b>
|
|
<p>
|
|
Suppose \( f \) is differentiable (i.e \( \nabla f(x) \) is well defined for
|
|
all \( x \) in the domain of \( f \)). Then \( f \) is convex if and only if \( D_f \)
|
|
is a convex set and <p> <br>
|
|
$$f(y) \geq f(x) + \nabla f(x)^T (y-x) $$
|
|
<p> <br> holds
|
|
for all \( x,y \in D_f \). This condition means that for a convex function
|
|
the first order Taylor expansion (right hand side above) at any point
|
|
a global under estimator of the function. To convince yourself you can
|
|
make a drawing of \( f(x) = x^2+1 \) and draw the tangent line to \( f(x) \) and
|
|
note that it is always below the graph.
|
|
</div>
|
|
|
|
<p>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b>Second order condition.</b>
|
|
<p>
|
|
Assume that \( f \) is twice
|
|
differentiable, i.e the Hessian matrix exists at each point in
|
|
\( D_f \). Then \( f \) is convex if and only if \( D_f \) is a convex set and its
|
|
Hessian is positive semi-definite for all \( x\in D_f \). For a
|
|
single-variable function this reduces to \( f''(x) \geq 0 \). Geometrically this means that \( f \) has nonnegative curvature
|
|
everywhere.
|
|
</div>
|
|
|
|
<p>
|
|
This condition is particularly useful since it gives us an procedure for determining if the function under consideration is convex, apart from using the definition.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec15">More on convex functions </h2>
|
|
|
|
<p>
|
|
The next result is of great importance to us and the reason why we are
|
|
going on about convex functions. In machine learning we frequently
|
|
have to minimize a loss/cost function in order to find the best
|
|
parameters for the model we are considering.
|
|
|
|
<p>
|
|
Ideally we want the
|
|
global minimum (for high-dimensional models it is hard to know
|
|
if we have local or global minimum). However, if the cost/loss function
|
|
is convex the following result provides invaluable information:
|
|
|
|
<p>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b>Any minimum is global for convex functions.</b>
|
|
<p>
|
|
Consider the problem of finding \( x \in \mathbb{R}^n \) such that \( f(x) \)
|
|
is minimal, where \( f \) is convex and differentiable. Then, any point
|
|
\( x^* \) that satisfies \( \nabla f(x^*) = 0 \) is a global minimum.
|
|
</div>
|
|
|
|
<p>
|
|
This result means that if we know that the cost/loss function is convex and we are able to find a minimum, we are guaranteed that it is a global minimum.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec16">Some simple problems </h2>
|
|
|
|
<ol>
|
|
<p><li> Show that \( f(x)=x^2 \) is convex for \( x \in \mathbb{R} \) using the definition of convexity. Hint: If you re-write the definition, \( f \) is convex if the following holds for all \( x,y \in D_f \) and any \( \lambda \in [0,1] \) $\lambda f(x)+(1-\lambda)f(y)-f(\lambda x + (1-\lambda) y ) \geq 0$.</li>
|
|
<p><li> Using the second order condition show that the following functions are convex on the specified domain.</li>
|
|
|
|
<ul>
|
|
<p><li> \( f(x) = e^x \) is convex for \( x \in \mathbb{R} \).</li>
|
|
<p><li> \( g(x) = -\ln(x) \) is convex for \( x \in (0,\infty) \).</li>
|
|
</ul>
|
|
<p><li> Let \( f(x) = x^2 \) and \( g(x) = e^x \). Show that \( f(g(x)) \) and \( g(f(x)) \) is convex for \( x \in \mathbb{R} \). Also show that if \( f(x) \) is any convex function than \( h(x) = e^{f(x)} \) is convex.</li>
|
|
<p><li> A norm is any function that satisfy the following properties</li>
|
|
|
|
<ul>
|
|
<p><li> \( f(\alpha x) = |\alpha| f(x) \) for all \( \alpha \in \mathbb{R} \).</li>
|
|
<p><li> \( f(x+y) \leq f(x) + f(y) \)</li>
|
|
<p><li> \( f(x) \leq 0 \) for all \( x \in \mathbb{R}^n \) with equality if and only if \( x = 0 \)</li>
|
|
</ul>
|
|
<p>
|
|
</ol>
|
|
<p>
|
|
|
|
Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this).
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec17">Standard steepest descent </h2>
|
|
|
|
<p>
|
|
Before we proceed, we would like to discuss the approach called the
|
|
<b>standard Steepest descent</b>, which again leads to us having to be able
|
|
to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG).
|
|
|
|
<p>
|
|
<a href="https://www.cs.cmu.edu/~quake-papers/painless-conjugate-gradient.pdf" target="_blank">The success of the CG method</a>
|
|
for finding solutions of non-linear problems is based on the theory
|
|
of conjugate gradients for linear systems of equations. It belongs to
|
|
the class of iterative methods for solving problems from linear
|
|
algebra of the type
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{A}\hat{x} = \hat{b}.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
In the iterative process we end up with a problem like
|
|
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{r}= \hat{b}-\hat{A}\hat{x},
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
where \( \hat{r} \) is the so-called residual or error in the iterative process.
|
|
|
|
<p>
|
|
When we have found the exact solution, \( \hat{r}=0 \).
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec18">Gradient method </h2>
|
|
|
|
<p>
|
|
The residual is zero when we reach the minimum of the quadratic equation
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b},
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
with the constraint that the matrix \( \hat{A} \) is positive definite and
|
|
symmetric. This defines also the Hessian and we want it to be positive definite.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec19">Steepest descent method </h2>
|
|
|
|
<p>
|
|
We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \).
|
|
We can assume without loss of generality that
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}_0=0,
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
or consider the system
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0,
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
instead.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec20">Steepest descent method </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition)
|
|
to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \),
|
|
which equals
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{A}\hat{x}_0-\hat{b},
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
and
|
|
\( \hat{x}_0=0 \) it is equal \( -\hat{b} \).
|
|
|
|
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec21">Final expressions </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
We can compute the residual iteratively as
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1},
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
which equals
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k),
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
or
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k,
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
which gives
|
|
|
|
<p> <br>
|
|
$$
|
|
\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k}
|
|
$$
|
|
<p> <br>
|
|
|
|
leading to the iterative scheme
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k},
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec22">Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
|
|
<!-- code=c++ (!bc cppcod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #1e889b">#include</span> <span style="color: #228B22"><cmath></span><span style="color: #1e889b"></span>
|
|
<span style="color: #1e889b">#include</span> <span style="color: #228B22"><iostream></span><span style="color: #1e889b"></span>
|
|
<span style="color: #1e889b">#include</span> <span style="color: #228B22"><fstream></span><span style="color: #1e889b"></span>
|
|
<span style="color: #1e889b">#include</span> <span style="color: #228B22"><iomanip></span><span style="color: #1e889b"></span>
|
|
<span style="color: #1e889b">#include</span> <span style="color: #228B22">"vectormatrixclass.h"</span><span style="color: #1e889b"></span>
|
|
<span style="color: #8B008B; font-weight: bold">using</span> <span style="color: #8B008B; font-weight: bold">namespace</span> std;
|
|
<span style="color: #228B22">// Main function begins here</span>
|
|
<span style="color: #00688B; font-weight: bold">int</span> <span style="color: #008b45">main</span>(<span style="color: #00688B; font-weight: bold">int</span> argc, <span style="color: #00688B; font-weight: bold">char</span> * argv[]){
|
|
<span style="color: #00688B; font-weight: bold">int</span> dim = <span style="color: #B452CD">2</span>;
|
|
Vector x(dim),xsd(dim), b(dim),x0(dim);
|
|
Matrix A(dim,dim);
|
|
|
|
<span style="color: #228B22">// Set our initial guess</span>
|
|
x0(<span style="color: #B452CD">0</span>) = x0(<span style="color: #B452CD">1</span>) = <span style="color: #B452CD">0</span>;
|
|
<span style="color: #228B22">// Set the matrix</span>
|
|
A(<span style="color: #B452CD">0</span>,<span style="color: #B452CD">0</span>) = <span style="color: #B452CD">3</span>; A(<span style="color: #B452CD">1</span>,<span style="color: #B452CD">0</span>) = <span style="color: #B452CD">2</span>; A(<span style="color: #B452CD">0</span>,<span style="color: #B452CD">1</span>) = <span style="color: #B452CD">2</span>; A(<span style="color: #B452CD">1</span>,<span style="color: #B452CD">1</span>) = <span style="color: #B452CD">6</span>;
|
|
b(<span style="color: #B452CD">0</span>) = <span style="color: #B452CD">2</span>; b(<span style="color: #B452CD">1</span>) = -<span style="color: #B452CD">8</span>;
|
|
cout << <span style="color: #CD5555">"The Matrix A that we are using: "</span> << endl;
|
|
A.Print();
|
|
cout << endl;
|
|
xsd = SteepestDescent(A,b,x0);
|
|
cout << <span style="color: #CD5555">"The approximate solution using Steepest Descent is: "</span> << endl;
|
|
xsd.Print();
|
|
cout << endl;
|
|
}
|
|
</pre></div>
|
|
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec23">The routine for the steepest descent method </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
|
|
<!-- code=c++ (!bc cppcod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span>Vector <span style="color: #008b45">SteepestDescent</span>(Matrix A, Vector b, Vector x0){
|
|
<span style="color: #00688B; font-weight: bold">int</span> IterMax, i;
|
|
<span style="color: #00688B; font-weight: bold">int</span> dim = x0.Dimension();
|
|
<span style="color: #8B008B; font-weight: bold">const</span> <span style="color: #00688B; font-weight: bold">double</span> tolerance = <span style="color: #B452CD">1.0e-14</span>;
|
|
Vector x(dim),f(dim),z(dim);
|
|
<span style="color: #00688B; font-weight: bold">double</span> c,alpha,d;
|
|
IterMax = <span style="color: #B452CD">30</span>;
|
|
x = x0;
|
|
r = A*x-b;
|
|
i = <span style="color: #B452CD">0</span>;
|
|
<span style="color: #8B008B; font-weight: bold">while</span> (i <= IterMax){
|
|
z = A*r;
|
|
c = dot(r,r);
|
|
alpha = c/dot(r,z);
|
|
x = x - alpha*r;
|
|
r = A*x-b;
|
|
<span style="color: #8B008B; font-weight: bold">if</span>(sqrt(dot(r,r)) < tolerance) <span style="color: #8B008B; font-weight: bold">break</span>;
|
|
i++;
|
|
}
|
|
<span style="color: #8B008B; font-weight: bold">return</span> x;
|
|
}
|
|
</pre></div>
|
|
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec24">Steepest descent example </h2>
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy.linalg</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">la</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">scipy.optimize</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">sopt</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">pt</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">mpl_toolkits.mplot3d</span> <span style="color: #8B008B; font-weight: bold">import</span> axes3d
|
|
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f</span>(x):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">0.5</span>*x[<span style="color: #B452CD">0</span>]**<span style="color: #B452CD">2</span> + <span style="color: #B452CD">2.5</span>*x[<span style="color: #B452CD">1</span>]**<span style="color: #B452CD">2</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">df</span>(x):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> np.array([x[<span style="color: #B452CD">0</span>], <span style="color: #B452CD">5</span>*x[<span style="color: #B452CD">1</span>]])
|
|
|
|
fig = pt.figure()
|
|
ax = fig.gca(projection=<span style="color: #CD5555">"3d"</span>)
|
|
|
|
xmesh, ymesh = np.mgrid[-<span style="color: #B452CD">2</span>:<span style="color: #B452CD">2</span>:<span style="color: #B452CD">50j</span>,-<span style="color: #B452CD">2</span>:<span style="color: #B452CD">2</span>:<span style="color: #B452CD">50j</span>]
|
|
fmesh = f(np.array([xmesh, ymesh]))
|
|
ax.plot_surface(xmesh, ymesh, fmesh)
|
|
</pre></div>
|
|
<p>
|
|
And then as countor plot
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span>pt.axis(<span style="color: #CD5555">"equal"</span>)
|
|
pt.contour(xmesh, ymesh, fmesh)
|
|
guesses = [np.array([<span style="color: #B452CD">2</span>, <span style="color: #B452CD">2.</span>/<span style="color: #B452CD">5</span>])]
|
|
</pre></div>
|
|
<p>
|
|
Find guesses
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span>x = guesses[-<span style="color: #B452CD">1</span>]
|
|
s = -df(x)
|
|
</pre></div>
|
|
<p>
|
|
Run it!
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f1d</span>(alpha):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> f(x + alpha*s)
|
|
|
|
alpha_opt = sopt.golden(f1d)
|
|
next_guess = x + alpha_opt * s
|
|
guesses.append(next_guess)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(next_guess)
|
|
</pre></div>
|
|
<p>
|
|
What happened?
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span>pt.axis(<span style="color: #CD5555">"equal"</span>)
|
|
pt.contour(xmesh, ymesh, fmesh, <span style="color: #B452CD">50</span>)
|
|
it_array = np.array(guesses)
|
|
pt.plot(it_array.T[<span style="color: #B452CD">0</span>], it_array.T[<span style="color: #B452CD">1</span>], <span style="color: #CD5555">"x-"</span>)
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec25">Conjugate gradient </h2>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec26">Revisiting our first homework </h2>
|
|
|
|
<p>
|
|
We will use linear regression as a case study for the gradient descent
|
|
methods. Linear regression is a great test case for the gradient
|
|
descent methods discussed in the lectures since it has several
|
|
desirable properties such as:
|
|
|
|
<ol>
|
|
<p><li> An analytical solution (recall homework set 1).</li>
|
|
<p><li> The gradient can be computed analytically.</li>
|
|
<p><li> The cost function is convex which guarantees that gradient descent converges for small enough learning rates</li>
|
|
</ol>
|
|
<p>
|
|
|
|
We revisit the example from homework set 1 where we had
|
|
<p> <br>
|
|
$$
|
|
y_i = 5x_i^2 + 0.1\xi_i, \ i=1,\cdots,100
|
|
$$
|
|
<p> <br>
|
|
|
|
with \( x_i \in [0,1] \) chosen randomly with a uniform distribution. Additionally \( \xi_i \) represents stochastic noise chosen according to a normal distribution \( \cal {N}(0,1) \).
|
|
The linear regression model is given by
|
|
<p> <br>
|
|
$$
|
|
h_\beta(x) = \hat{y} = \beta_0 + \beta_1 x,
|
|
$$
|
|
<p> <br>
|
|
|
|
such that
|
|
<p> <br>
|
|
$$
|
|
\hat{y}_i = \beta_0 + \beta_1 x_i.
|
|
$$
|
|
<p> <br>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec27">Gradient descent example </h2>
|
|
|
|
<p>
|
|
Let \( \mathbf{y} = (y_1,\cdots,y_n)^T \), \( \mathbf{\hat{y}} = (\hat{y}_1,\cdots,\hat{y}_n)^T \) and \( \beta = (\beta_0, \beta_1)^T \)
|
|
|
|
<p>
|
|
It is convenient to write \( \mathbf{\hat{y}} = X\beta \) where \( X \in \mathbb{R}^{100 \times 2} \) is the design matrix given by
|
|
<p> <br>
|
|
$$
|
|
X \equiv \begin{bmatrix}
|
|
1 & x_1 \\
|
|
\vdots & \vdots \\
|
|
1 & x_{100} & \\
|
|
\end{bmatrix}.
|
|
$$
|
|
<p> <br>
|
|
|
|
The loss function is given by
|
|
<p> <br>
|
|
$$
|
|
C(\beta) = ||X\beta-\mathbf{y}||^2 = ||X\beta||^2 - 2 \mathbf{y}^T X\beta + ||\mathbf{y}||^2 = \sum_{i=1}^{100} (\beta_0 + \beta_1 x_i)^2 - 2 y_i (\beta_0 + \beta_1 x_i) + y_i^2
|
|
$$
|
|
<p> <br>
|
|
|
|
and we want to find \( \beta \) such that \( C(\beta) \) is minimized.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec28">The derivative of the cost/loss function </h2>
|
|
|
|
<p>
|
|
Computing \( \partial C(\beta) / \partial \beta_0 \) and \( \partial C(\beta) / \partial \beta_1 \) we can show that the gradient can be written as
|
|
<p> <br>
|
|
$$
|
|
\nabla_{\beta} C(\beta) = (\partial C(\beta) / \partial \beta_0, \partial C(\beta) / \partial \beta_1)^T = 2\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\
|
|
\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\
|
|
\end{bmatrix} = 2X^T(X\beta - \mathbf{y}),
|
|
$$
|
|
<p> <br>
|
|
|
|
where \( X \) is the design matrix defined above.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec29">The Hessian matrix </h2>
|
|
The Hessian matrix of \( C(\beta) \) is given by
|
|
<p> <br>
|
|
$$
|
|
\hat{H} \equiv \begin{bmatrix}
|
|
\frac{\partial^2 C(\beta)}{\partial \beta_0^2} & \frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} \\
|
|
\frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} & \frac{\partial^2 C(\beta)}{\partial \beta_1^2} & \\
|
|
\end{bmatrix} = 2X^T X.
|
|
$$
|
|
<p> <br>
|
|
|
|
This result implies that \( C(\beta) \) is a convex function since the matrix \( X^T X \) always is positive semi-definite.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec30">Simple program </h2>
|
|
|
|
<p>
|
|
We can now write a program that minimizes \( C(\beta) \) using the gradient descent method with a constant learning rate \( \gamma \) according to
|
|
<p> <br>
|
|
$$
|
|
\beta_{k+1} = \beta_k - \gamma \nabla_\beta C(\beta_k), \ k=0,1,\cdots
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
We can use the expression we computed for the gradient and let use a
|
|
\( \beta_0 \) be chosen randomly and let \( \gamma = 0.001 \). Stop iterating
|
|
when \( ||\nabla_\beta C(\beta_k) || \leq \epsilon = 10^{-8} \).
|
|
|
|
<p>
|
|
And finally we can compare our solution for \( \beta \) with the analytic result given by
|
|
\( \beta= (X^TX)^{-1} X^T \mathbf{y} \).
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
|
|
<span style="color: #CD5555">"""</span>
|
|
<span style="color: #CD5555">The following setup is just a suggestion, feel free to write it the way you like.</span>
|
|
<span style="color: #CD5555">"""</span>
|
|
|
|
<span style="color: #228B22">#Setup problem described in the exercise</span>
|
|
N = <span style="color: #B452CD">100</span> <span style="color: #228B22">#Nr of datapoints</span>
|
|
M = <span style="color: #B452CD">2</span> <span style="color: #228B22">#Nr of features</span>
|
|
x = np.random.rand(N) <span style="color: #228B22">#Uniformly generated x-values in [0,1]</span>
|
|
y = <span style="color: #B452CD">5</span>*x**<span style="color: #B452CD">2</span> + <span style="color: #B452CD">0.1</span>*np.random.randn(N)
|
|
X = np.c_[np.ones(N),x] <span style="color: #228B22">#Construct design matrix</span>
|
|
|
|
<span style="color: #228B22">#Compute beta according to normal equations to compare with GD solution</span>
|
|
Xt_X_inv = np.linalg.inv(np.dot(X.T,X))
|
|
Xt_y = np.dot(X.transpose(),y)
|
|
beta_NE = np.dot(Xt_X_inv,Xt_y)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(beta_NE)
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec31">Gradient Descent Example </h2>
|
|
|
|
<p>
|
|
Another simple example is here
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #228B22"># Importing various packages</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">random</span> <span style="color: #8B008B; font-weight: bold">import</span> random, seed
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">mpl_toolkits.mplot3d</span> <span style="color: #8B008B; font-weight: bold">import</span> Axes3D
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">matplotlib</span> <span style="color: #8B008B; font-weight: bold">import</span> cm
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">matplotlib.ticker</span> <span style="color: #8B008B; font-weight: bold">import</span> LinearLocator, FormatStrFormatter
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">sys</span>
|
|
|
|
x = <span style="color: #B452CD">2</span>*np.random.rand(<span style="color: #B452CD">100</span>,<span style="color: #B452CD">1</span>)
|
|
y = <span style="color: #B452CD">4</span>+<span style="color: #B452CD">3</span>*x+np.random.randn(<span style="color: #B452CD">100</span>,<span style="color: #B452CD">1</span>)
|
|
|
|
xb = np.c_[np.ones((<span style="color: #B452CD">100</span>,<span style="color: #B452CD">1</span>)), x]
|
|
beta_linreg = np.linalg.inv(xb.T.dot(xb)).dot(xb.T).dot(y)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(beta_linreg)
|
|
beta = np.random.randn(<span style="color: #B452CD">2</span>,<span style="color: #B452CD">1</span>)
|
|
|
|
eta = <span style="color: #B452CD">0.1</span>
|
|
Niterations = <span style="color: #B452CD">1000</span>
|
|
m = <span style="color: #B452CD">100</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">for</span> <span style="color: #658b00">iter</span> <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(Niterations):
|
|
gradients = <span style="color: #B452CD">2.0</span>/m*xb.T.dot(xb.dot(beta)-y)
|
|
beta -= eta*gradients
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(beta)
|
|
xnew = np.array([[<span style="color: #B452CD">0</span>],[<span style="color: #B452CD">2</span>]])
|
|
xbnew = np.c_[np.ones((<span style="color: #B452CD">2</span>,<span style="color: #B452CD">1</span>)), xnew]
|
|
ypredict = xbnew.dot(beta)
|
|
ypredict2 = xbnew.dot(beta_linreg)
|
|
plt.plot(xnew, ypredict, <span style="color: #CD5555">"r-"</span>)
|
|
plt.plot(xnew, ypredict2, <span style="color: #CD5555">"b-"</span>)
|
|
plt.plot(x, y ,<span style="color: #CD5555">'ro'</span>)
|
|
plt.axis([<span style="color: #B452CD">0</span>,<span style="color: #B452CD">2.0</span>,<span style="color: #B452CD">0</span>, <span style="color: #B452CD">15.0</span>])
|
|
plt.xlabel(<span style="color: #CD5555">r'$x$'</span>)
|
|
plt.ylabel(<span style="color: #CD5555">r'$y$'</span>)
|
|
plt.title(<span style="color: #CD5555">r'Gradient descent example'</span>)
|
|
plt.show()
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec32">And a corresponding example using <b>scikit-learn</b> </h2>
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #228B22"># Importing various packages</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">random</span> <span style="color: #8B008B; font-weight: bold">import</span> random, seed
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> SGDRegressor
|
|
|
|
x = <span style="color: #B452CD">2</span>*np.random.rand(<span style="color: #B452CD">100</span>,<span style="color: #B452CD">1</span>)
|
|
y = <span style="color: #B452CD">4</span>+<span style="color: #B452CD">3</span>*x+np.random.randn(<span style="color: #B452CD">100</span>,<span style="color: #B452CD">1</span>)
|
|
|
|
xb = np.c_[np.ones((<span style="color: #B452CD">100</span>,<span style="color: #B452CD">1</span>)), x]
|
|
beta_linreg = np.linalg.inv(xb.T.dot(xb)).dot(xb.T).dot(y)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(beta_linreg)
|
|
sgdreg = SGDRegressor(n_iter = <span style="color: #B452CD">50</span>, penalty=<span style="color: #658b00">None</span>, eta0=<span style="color: #B452CD">0.1</span>)
|
|
sgdreg.fit(x,y.ravel())
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(sgdreg.intercept_, sgdreg.coef_)
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec33">Gradient descent and Ridge </h2>
|
|
|
|
<p>
|
|
We have also discussed Ridge regression where the loss function contains a regularized given by the \( L_2 \) norm of \( \beta \),
|
|
<p> <br>
|
|
$$
|
|
C_{\text{ridge}}(\beta) = ||X\beta -\mathbf{y}||^2 + \lambda ||\beta||^2, \ \lambda \geq 0.
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
In order to minimize \( C_{\text{ridge}}(\beta) \) using GD we only have adjust the gradient as follows
|
|
<p> <br>
|
|
$$
|
|
\nabla_\beta C_{\text{ridge}}(\beta) = 2\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\
|
|
\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\
|
|
\end{bmatrix} + 2\lambda\begin{bmatrix} \beta_0 \\ \beta_1\end{bmatrix} = 2 (X^T(X\beta - \mathbf{y})+\lambda \beta).
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
We can now extend our program to minimize \( C_{\text{ridge}}(\beta) \) using gradient descent and compare with the analytical solution given by
|
|
<p> <br>
|
|
$$
|
|
\beta_{\text{ridge}} = \left(X^T X + \lambda I_{2 \times 2} \right)^{-1} X^T \mathbf{y},
|
|
$$
|
|
<p> <br>
|
|
|
|
for \( \lambda = {0,1,10,50,100} \) (\( \lambda = 0 \) corresponds to ordinary least squares).
|
|
We can then compute \( ||\beta_{\text{ridge}}|| \) for each \( \lambda \).
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
|
|
<span style="color: #CD5555">"""</span>
|
|
<span style="color: #CD5555">The following setup is just a suggestion, feel free to write it the way you like.</span>
|
|
<span style="color: #CD5555">"""</span>
|
|
|
|
<span style="color: #228B22">#Setup problem described in the exercise</span>
|
|
N = <span style="color: #B452CD">100</span> <span style="color: #228B22">#Nr of datapoints</span>
|
|
M = <span style="color: #B452CD">2</span> <span style="color: #228B22">#Nr of features</span>
|
|
x = np.random.rand(N)
|
|
y = <span style="color: #B452CD">5</span>*x**<span style="color: #B452CD">2</span> + <span style="color: #B452CD">0.1</span>*np.random.randn(N)
|
|
|
|
|
|
<span style="color: #228B22">#Compute analytic beta for Ridge regression </span>
|
|
X = np.c_[np.ones(N),x]
|
|
XT_X = np.dot(X.T,X)
|
|
|
|
l = <span style="color: #B452CD">0.1</span> <span style="color: #228B22">#Ridge parameter lambda</span>
|
|
Id = np.eye(XT_X.shape[<span style="color: #B452CD">0</span>])
|
|
|
|
Z = np.linalg.inv(XT_X+l*Id)
|
|
beta_ridge = np.dot(Z,np.dot(X.T,y))
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(beta_ridge)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(np.linalg.norm(beta_ridge)) <span style="color: #228B22">#||beta||</span>
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec34">Automatic differentiation </h2>
|
|
Python has tools for so-called <b>automatic differentiation</b>.
|
|
Consider the following example
|
|
<p> <br>
|
|
$$
|
|
f(x) = \sin\left(2\pi x + x^2\right)
|
|
$$
|
|
<p> <br>
|
|
|
|
which has the following derivative
|
|
<p> <br>
|
|
$$
|
|
f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right)
|
|
$$
|
|
<p> <br>
|
|
|
|
Using <b>autograd</b> we have
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
|
|
<span style="color: #228B22"># To do elementwise differentiation:</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> elementwise_grad <span style="color: #8B008B; font-weight: bold">as</span> egrad
|
|
|
|
<span style="color: #228B22"># To plot:</span>
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
|
|
|
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f</span>(x):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> np.sin(<span style="color: #B452CD">2</span>*np.pi*x + x**<span style="color: #B452CD">2</span>)
|
|
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f_grad_analytic</span>(x):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> np.cos(<span style="color: #B452CD">2</span>*np.pi*x + x**<span style="color: #B452CD">2</span>)*(<span style="color: #B452CD">2</span>*np.pi + <span style="color: #B452CD">2</span>*x)
|
|
|
|
<span style="color: #228B22"># Do the comparison:</span>
|
|
x = np.linspace(<span style="color: #B452CD">0</span>,<span style="color: #B452CD">1</span>,<span style="color: #B452CD">1000</span>)
|
|
|
|
f_grad = egrad(f)
|
|
|
|
computed = f_grad(x)
|
|
analytic = f_grad_analytic(x)
|
|
|
|
plt.title(<span style="color: #CD5555">'Derivative computed from Autograd compared with the analytical derivative'</span>)
|
|
plt.plot(x,computed,label=<span style="color: #CD5555">'autograd'</span>)
|
|
plt.plot(x,analytic,label=<span style="color: #CD5555">'analytic'</span>)
|
|
|
|
plt.xlabel(<span style="color: #CD5555">'x'</span>)
|
|
plt.ylabel(<span style="color: #CD5555">'y'</span>)
|
|
plt.legend()
|
|
|
|
plt.show()
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The max absolute difference is: %g"</span>%(np.max(np.abs(computed - analytic))))
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec35">Using autograd </h2>
|
|
|
|
<p>
|
|
Here we
|
|
experiment with what kind of functions Autograd is capable
|
|
of finding the gradient of. The following Python functions are just
|
|
meant to illustrate what Autograd can do, but please feel free to
|
|
experiment with other, possibly more complicated, functions as well.
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f1</span>(x):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> x**<span style="color: #B452CD">3</span> + <span style="color: #B452CD">1</span>
|
|
|
|
f1_grad = grad(f1)
|
|
|
|
<span style="color: #228B22"># Remember to send in float as argument to the computed gradient from Autograd!</span>
|
|
a = <span style="color: #B452CD">1.0</span>
|
|
|
|
<span style="color: #228B22"># See the evaluated gradient at a using autograd:</span>
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The gradient of f1 evaluated at a = %g using autograd is: %g"</span>%(a,f1_grad(a)))
|
|
|
|
<span style="color: #228B22"># Compare with the analytical derivative, that is f1'(x) = 3*x**2 </span>
|
|
grad_analytical = <span style="color: #B452CD">3</span>*a**<span style="color: #B452CD">2</span>
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"</span>%(a,grad_analytical))
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec36">Autograd with more complicated functions </h2>
|
|
|
|
<p>
|
|
To differentiate with respect to two (or more) arguments of a Python
|
|
function, Autograd need to know at which variable the function if
|
|
being differentiated with respect to.
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f2</span>(x1,x2):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">3</span>*x1**<span style="color: #B452CD">3</span> + x2*(x1 - <span style="color: #B452CD">5</span>) + <span style="color: #B452CD">1</span>
|
|
|
|
<span style="color: #228B22"># By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1</span>
|
|
f2_grad_x1 = grad(f2,<span style="color: #B452CD">0</span>)
|
|
|
|
<span style="color: #228B22"># ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad</span>
|
|
f2_grad_x2 = grad(f2,<span style="color: #B452CD">1</span>)
|
|
|
|
x1 = <span style="color: #B452CD">1.0</span>
|
|
x2 = <span style="color: #B452CD">3.0</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"Evaluating at x1 = %g, x2 = %g"</span>%(x1,x2))
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"-"</span>*<span style="color: #B452CD">30</span>)
|
|
|
|
<span style="color: #228B22"># Compare with the analytical derivatives:</span>
|
|
|
|
<span style="color: #228B22"># Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:</span>
|
|
f2_grad_x1_analytical = <span style="color: #B452CD">9</span>*x1**<span style="color: #B452CD">2</span> + x2
|
|
|
|
<span style="color: #228B22"># Derivative of f2 w.r.t x2 is: x1 - 5:</span>
|
|
f2_grad_x2_analytical = x1 - <span style="color: #B452CD">5</span>
|
|
|
|
<span style="color: #228B22"># See the evaluated derivations:</span>
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The derivative of f2 w.r.t x1: %g"</span>%( f2_grad_x1(x1,x2) ))
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The analytical derivative of f2 w.r.t x1: %g"</span>%( f2_grad_x1(x1,x2) ))
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>()
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The derivative of f2 w.r.t x2: %g"</span>%( f2_grad_x2(x1,x2) ))
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The analytical derivative of f2 w.r.t x2: %g"</span>%( f2_grad_x2(x1,x2) ))
|
|
</pre></div>
|
|
<p>
|
|
Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec37">More complicated functions using the elements of their arguments directly </h2>
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f3</span>(x): <span style="color: #228B22"># Assumes x is an array of length 5 or higher</span>
|
|
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">2</span>*x[<span style="color: #B452CD">0</span>] + <span style="color: #B452CD">3</span>*x[<span style="color: #B452CD">1</span>] + <span style="color: #B452CD">5</span>*x[<span style="color: #B452CD">2</span>] + <span style="color: #B452CD">7</span>*x[<span style="color: #B452CD">3</span>] + <span style="color: #B452CD">11</span>*x[<span style="color: #B452CD">4</span>]**<span style="color: #B452CD">2</span>
|
|
|
|
f3_grad = grad(f3)
|
|
|
|
x = np.linspace(<span style="color: #B452CD">0</span>,<span style="color: #B452CD">4</span>,<span style="color: #B452CD">5</span>)
|
|
|
|
<span style="color: #228B22"># Print the computed gradient:</span>
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The computed gradient of f3 is: "</span>, f3_grad(x))
|
|
|
|
<span style="color: #228B22"># The analytical gradient is: (2, 3, 5, 7, 22*x[4])</span>
|
|
f3_grad_analytical = np.array([<span style="color: #B452CD">2</span>, <span style="color: #B452CD">3</span>, <span style="color: #B452CD">5</span>, <span style="color: #B452CD">7</span>, <span style="color: #B452CD">22</span>*x[<span style="color: #B452CD">4</span>]])
|
|
|
|
<span style="color: #228B22"># Print the analytical gradient:</span>
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The analytical gradient of f3 is: "</span>, f3_grad_analytical)
|
|
</pre></div>
|
|
<p>
|
|
Note that in this case, when sending an array as input argument, the
|
|
output from Autograd is another array. This is the true gradient of
|
|
the function, as opposed to the function in the previous example. By
|
|
using arrays to represent the variables, the output from Autograd
|
|
might be easier to work with, as the output is closer to what one
|
|
could expect form a gradient-evaluting function.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec38">Functions using mathematical functions from Numpy </h2>
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f4</span>(x):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> np.sqrt(<span style="color: #B452CD">1</span>+x**<span style="color: #B452CD">2</span>) + np.exp(x) + np.sin(<span style="color: #B452CD">2</span>*np.pi*x)
|
|
|
|
f4_grad = grad(f4)
|
|
|
|
x = <span style="color: #B452CD">2.7</span>
|
|
|
|
<span style="color: #228B22"># Print the computed derivative:</span>
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The computed derivative of f4 at x = %g is: %g"</span>%(x,f4_grad(x)))
|
|
|
|
<span style="color: #228B22"># The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi</span>
|
|
f4_grad_analytical = x/np.sqrt(<span style="color: #B452CD">1</span> + x**<span style="color: #B452CD">2</span>) + np.exp(x) + np.cos(<span style="color: #B452CD">2</span>*np.pi*x)*<span style="color: #B452CD">2</span>*np.pi
|
|
|
|
<span style="color: #228B22"># Print the analytical gradient:</span>
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The analytical gradient of f4 at x = %g is: %g"</span>%(x,f4_grad_analytical))
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec39">More autograd </h2>
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f5</span>(x):
|
|
<span style="color: #8B008B; font-weight: bold">if</span> x >= <span style="color: #B452CD">0</span>:
|
|
<span style="color: #8B008B; font-weight: bold">return</span> x**<span style="color: #B452CD">2</span>
|
|
<span style="color: #8B008B; font-weight: bold">else</span>:
|
|
<span style="color: #8B008B; font-weight: bold">return</span> -<span style="color: #B452CD">3</span>*x + <span style="color: #B452CD">1</span>
|
|
|
|
f5_grad = grad(f5)
|
|
|
|
x = <span style="color: #B452CD">2.7</span>
|
|
|
|
<span style="color: #228B22"># Print the computed derivative:</span>
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The computed derivative of f5 at x = %g is: %g"</span>%(x,f5_grad(x)))
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec40">And with loops </h2>
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f6_for</span>(x):
|
|
val = <span style="color: #B452CD">0</span>
|
|
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">10</span>):
|
|
val = val + x**i
|
|
<span style="color: #8B008B; font-weight: bold">return</span> val
|
|
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f6_while</span>(x):
|
|
val = <span style="color: #B452CD">0</span>
|
|
i = <span style="color: #B452CD">0</span>
|
|
<span style="color: #8B008B; font-weight: bold">while</span> i < <span style="color: #B452CD">10</span>:
|
|
val = val + x**i
|
|
i = i + <span style="color: #B452CD">1</span>
|
|
<span style="color: #8B008B; font-weight: bold">return</span> val
|
|
|
|
f6_for_grad = grad(f6_for)
|
|
f6_while_grad = grad(f6_while)
|
|
|
|
x = <span style="color: #B452CD">0.5</span>
|
|
|
|
<span style="color: #228B22"># Print the computed derivaties of f6_for and f6_while</span>
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The computed derivative of f6_for at x = %g is: %g"</span>%(x,f6_for_grad(x)))
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The computed derivative of f6_while at x = %g is: %g"</span>%(x,f6_while_grad(x)))
|
|
</pre></div>
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
<span style="color: #228B22"># Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9</span>
|
|
<span style="color: #228B22"># The analytical derivative is: sum(i*x**(i-1)) </span>
|
|
f6_grad_analytical = <span style="color: #B452CD">0</span>
|
|
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">10</span>):
|
|
f6_grad_analytical += i*x**(i-<span style="color: #B452CD">1</span>)
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The analytical derivative of f6 at x = %g is: %g"</span>%(x,f6_grad_analytical))
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec41">Using recursion </h2>
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f7</span>(n): <span style="color: #228B22"># Assume that n is an integer</span>
|
|
<span style="color: #8B008B; font-weight: bold">if</span> n == <span style="color: #B452CD">1</span> <span style="color: #8B008B">or</span> n == <span style="color: #B452CD">0</span>:
|
|
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">1</span>
|
|
<span style="color: #8B008B; font-weight: bold">else</span>:
|
|
<span style="color: #8B008B; font-weight: bold">return</span> n*f7(n-<span style="color: #B452CD">1</span>)
|
|
|
|
f7_grad = grad(f7)
|
|
|
|
n = <span style="color: #B452CD">2.0</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The computed derivative of f7 at n = %d is: %g"</span>%(n,f7_grad(n)))
|
|
|
|
<span style="color: #228B22"># The function f7 is an implementation of the factorial of n.</span>
|
|
<span style="color: #228B22"># By using the product rule, one can find that the derivative is:</span>
|
|
|
|
f7_grad_analytical = <span style="color: #B452CD">0</span>
|
|
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #658b00">int</span>(n)-<span style="color: #B452CD">1</span>):
|
|
tmp = <span style="color: #B452CD">1</span>
|
|
<span style="color: #8B008B; font-weight: bold">for</span> k <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #658b00">int</span>(n)-<span style="color: #B452CD">1</span>):
|
|
<span style="color: #8B008B; font-weight: bold">if</span> k != i:
|
|
tmp *= (n - k)
|
|
f7_grad_analytical += tmp
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The analytical derivative of f7 at n = %d is: %g"</span>%(n,f7_grad_analytical))
|
|
</pre></div>
|
|
<p>
|
|
Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec42">Unsupported functions </h2>
|
|
Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd.
|
|
|
|
<p>
|
|
Assigning a value to the variable being differentiated with respect to
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f8</span>(x): <span style="color: #228B22"># Assume x is an array</span>
|
|
x[<span style="color: #B452CD">2</span>] = <span style="color: #B452CD">3</span>
|
|
<span style="color: #8B008B; font-weight: bold">return</span> x*<span style="color: #B452CD">2</span>
|
|
|
|
f8_grad = grad(f8)
|
|
|
|
x = <span style="color: #B452CD">8.4</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The derivative of f8 is:"</span>,f8_grad(x))
|
|
</pre></div>
|
|
<p>
|
|
Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec43">The syntax a.dot(b) when finding the dot product </h2>
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f9</span>(a): <span style="color: #228B22"># Assume a is an array with 2 elements</span>
|
|
b = np.array([<span style="color: #B452CD">1.0</span>,<span style="color: #B452CD">2.0</span>])
|
|
<span style="color: #8B008B; font-weight: bold">return</span> a.dot(b)
|
|
|
|
f9_grad = grad(f9)
|
|
|
|
x = np.array([<span style="color: #B452CD">1.0</span>,<span style="color: #B452CD">0.0</span>])
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The derivative of f9 is:"</span>,f9_grad(x))
|
|
</pre></div>
|
|
<p>
|
|
Here we are told that the 'dot' function does not belong to Autograd's
|
|
version of a Numpy array. To overcome this, an alternative syntax
|
|
which also computed the dot product can be used:
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">autograd.numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">autograd</span> <span style="color: #8B008B; font-weight: bold">import</span> grad
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">f9_alternative</span>(x): <span style="color: #228B22"># Assume a is an array with 2 elements</span>
|
|
b = np.array([<span style="color: #B452CD">1.0</span>,<span style="color: #B452CD">2.0</span>])
|
|
<span style="color: #8B008B; font-weight: bold">return</span> np.dot(x,b) <span style="color: #228B22"># The same as x_1*b_1 + x_2*b_2</span>
|
|
|
|
f9_alternative_grad = grad(f9_alternative)
|
|
|
|
x = np.array([<span style="color: #B452CD">3.0</span>,<span style="color: #B452CD">0.0</span>])
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"The gradient of f9 is:"</span>,f9_alternative_grad(x))
|
|
|
|
<span style="color: #228B22"># The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively</span>
|
|
<span style="color: #228B22"># w.r.t x is (b_1, b_2).</span>
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec44">Recommended to avoid </h2>
|
|
The documentation recommends to avoid inplace operations such as
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span>a += b
|
|
a -= b
|
|
a*= b
|
|
a /=b
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec45">Stochastic Gradient Descent </h2>
|
|
|
|
<p>
|
|
Stochastic gradient descent (SGD) and variants thereof address some of
|
|
the shortcomings of the Gradient descent method discussed above.
|
|
|
|
<p>
|
|
The underlying idea of SGD comes from the observation that the cost
|
|
function, which we want to minimize, can almost always be written as a
|
|
sum over \( n \) data points \( \{\mathbf{x}_i\}_{i=1}^n \),
|
|
<p> <br>
|
|
$$
|
|
C(\mathbf{\beta}) = \sum_{i=1}^n c_i(\mathbf{x}_i,
|
|
\mathbf{\beta}).
|
|
$$
|
|
<p> <br>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec46">Computation of gradients </h2>
|
|
|
|
<p>
|
|
This in turn means that the gradient can be
|
|
computed as a sum over \( i \)-gradients
|
|
<p> <br>
|
|
$$
|
|
\nabla_\beta C(\mathbf{\beta}) = \sum_i^n \nabla_\beta c_i(\mathbf{x}_i,
|
|
\mathbf{\beta}).
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
Stochasticity/randomness is introduced by only taking the
|
|
gradient on a subset of the data called minibatches. If there are \( n \)
|
|
data points and the size of each minibatch is \( M \), there will be \( n/M \)
|
|
minibatches. We denote these minibatches by \( B_k \) where
|
|
\( k=1,\cdots,n/M \).
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec47">SGD example </h2>
|
|
As an example, suppose we have \( 10 \) data points \( (\mathbf{x}_1,\cdots, \mathbf{x}_{10}) \)
|
|
and we choose to have \( M=5 \) minibathces,
|
|
then each minibatch contains two data points. In particular we have
|
|
\( B_1 = (\mathbf{x}_1,\mathbf{x}_2), \cdots, B_5 =
|
|
(\mathbf{x}_9,\mathbf{x}_{10}) \). Note that if you choose \( M=1 \) you
|
|
have only a single batch with all data points and on the other extreme,
|
|
you may choose \( M=n \) resulting in a minibatch for each datapoint, i.e
|
|
\( B_k = \mathbf{x}_k \).
|
|
|
|
<p>
|
|
The idea is now to approximate the gradient by replacing the sum over
|
|
all data points with a sum over the data points in one the minibatches
|
|
picked at random in each gradient descent step
|
|
<p> <br>
|
|
$$
|
|
\nabla_{\beta}
|
|
C(\mathbf{\beta}) = \sum_{i=1}^n \nabla_\beta c_i(\mathbf{x}_i,
|
|
\mathbf{\beta}) \rightarrow \sum_{i \in B_k}^n \nabla_\beta
|
|
c_i(\mathbf{x}_i, \mathbf{\beta}).
|
|
$$
|
|
<p> <br>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec48">The gradient step </h2>
|
|
|
|
<p>
|
|
Thus a gradient descent step now looks like
|
|
<p> <br>
|
|
$$
|
|
\beta_{j+1} = \beta_j - \gamma_j \sum_{i \in B_k}^n \nabla_\beta c_i(\mathbf{x}_i,
|
|
\mathbf{\beta})
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
where \( k \) is picked at random with equal
|
|
probability from \( [1,n/M] \). An iteration over the number of
|
|
minibathces (n/M) is commonly referred to as an epoch. Thus it is
|
|
typical to choose a number of epochs and for each epoch iterate over
|
|
the number of minibatches, as exemplified in the code below.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec49">Simple example code </h2>
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
|
|
n = <span style="color: #B452CD">100</span> <span style="color: #228B22">#100 datapoints </span>
|
|
M = <span style="color: #B452CD">5</span> <span style="color: #228B22">#size of each minibatch</span>
|
|
m = <span style="color: #658b00">int</span>(n/M) <span style="color: #228B22">#number of minibatches</span>
|
|
n_epochs = <span style="color: #B452CD">10</span> <span style="color: #228B22">#number of epochs</span>
|
|
|
|
j = <span style="color: #B452CD">0</span>
|
|
<span style="color: #8B008B; font-weight: bold">for</span> epoch <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>,n_epochs+<span style="color: #B452CD">1</span>):
|
|
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(m):
|
|
k = np.random.randint(m) <span style="color: #228B22">#Pick the k-th minibatch at random</span>
|
|
<span style="color: #228B22">#Compute the gradient using the data in minibatch Bk</span>
|
|
<span style="color: #228B22">#Compute new suggestion for </span>
|
|
j += <span style="color: #B452CD">1</span>
|
|
</pre></div>
|
|
<p>
|
|
Taking the gradient only on a subset of the data has two important
|
|
benefits. First, it introduces randomness which decreases the chance
|
|
that our opmization scheme gets stuck in a local minima. Second, if
|
|
the size of the minibatches are small relative to the number of
|
|
datapoints (\( M < n \)), the computation of the gradient is much
|
|
cheaper since we sum over the datapoints in the \( k-th \) minibatch and not
|
|
all \( n \) datapoints.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec50">When do we stop? </h2>
|
|
|
|
<p>
|
|
A natural question is when do we stop the search for a new minimum?
|
|
One possibility is to compute the full gradient after a given number
|
|
of epochs and check if the norm of the gradient is smaller than some
|
|
threshold and stop if true. However, the condition that the gradient
|
|
is zero is valid also for local minima, so this would only tell us
|
|
that we are close to a local/global minimum. However, we could also
|
|
evaluate the cost function at this point, store the result and
|
|
continue the search. If the test kicks in at a later stage we can
|
|
compare the values of the cost function and keep the \( \beta \) that
|
|
gave the lowest value.
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec51">Slightly different approach </h2>
|
|
|
|
<p>
|
|
Another approach is to let the step length \( \gamma_j \) depend on the
|
|
number of epochs in such a way that it becomes very small after a
|
|
reasonable time such that we do not move at all.
|
|
|
|
<p>
|
|
As an example, let \( e = 0,1,2,3,\cdots \) denote the current epoch and let \( t_0, t_1 > 0 \) be two fixed numbers. Furthermore, let \( t = e \cdot m + i \) where \( m \) is the number of minibatches and \( i=0,\cdots,m-1 \). Then the function <p> <br>
|
|
$$\gamma_j(t; t_0, t_1) = \frac{t_0}{t+t_1} $$
|
|
<p> <br> goes to zero as the number of epochs gets large. I.e. we start with a step length \( \gamma_j (0; t_0, t_1) = t_0/t_1 \) which decays in <em>time</em> \( t \).
|
|
|
|
<p>
|
|
In this way we can fix the number of epochs, compute \( \beta \) and
|
|
evaluate the cost function at the end. Repeating the computation will
|
|
give a different result since the scheme is random by design. Then we
|
|
pick the final \( \beta \) that gives the lowest value of the cost
|
|
function.
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">step_length</span>(t,t0,t1):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> t0/(t+t1)
|
|
|
|
n = <span style="color: #B452CD">100</span> <span style="color: #228B22">#100 datapoints </span>
|
|
M = <span style="color: #B452CD">5</span> <span style="color: #228B22">#size of each minibatch</span>
|
|
m = <span style="color: #658b00">int</span>(n/M) <span style="color: #228B22">#number of minibatches</span>
|
|
n_epochs = <span style="color: #B452CD">500</span> <span style="color: #228B22">#number of epochs</span>
|
|
t0 = <span style="color: #B452CD">1.0</span>
|
|
t1 = <span style="color: #B452CD">10</span>
|
|
|
|
gamma_j = t0/t1
|
|
j = <span style="color: #B452CD">0</span>
|
|
<span style="color: #8B008B; font-weight: bold">for</span> epoch <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>,n_epochs+<span style="color: #B452CD">1</span>):
|
|
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(m):
|
|
k = np.random.randint(m) <span style="color: #228B22">#Pick the k-th minibatch at random</span>
|
|
<span style="color: #228B22">#Compute the gradient using the data in minibatch Bk</span>
|
|
<span style="color: #228B22">#Compute new suggestion for beta</span>
|
|
t = epoch*m+i
|
|
gamma_j = step_length(t,t0,t1)
|
|
j += <span style="color: #B452CD">1</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"gamma_j after %d epochs: %g"</span> % (n_epochs,gamma_j))
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec52">Program for stochastic gradient </h2>
|
|
|
|
<p>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span><span style="color: #228B22"># Importing various packages</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">math</span> <span style="color: #8B008B; font-weight: bold">import</span> exp, sqrt
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">random</span> <span style="color: #8B008B; font-weight: bold">import</span> random, seed
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
|
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
|
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> SGDRegressor
|
|
|
|
x = <span style="color: #B452CD">2</span>*np.random.rand(<span style="color: #B452CD">100</span>,<span style="color: #B452CD">1</span>)
|
|
y = <span style="color: #B452CD">4</span>+<span style="color: #B452CD">3</span>*x+np.random.randn(<span style="color: #B452CD">100</span>,<span style="color: #B452CD">1</span>)
|
|
|
|
xb = np.c_[np.ones((<span style="color: #B452CD">100</span>,<span style="color: #B452CD">1</span>)), x]
|
|
theta_linreg = np.linalg.inv(xb.T.dot(xb)).dot(xb.T).dot(y)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"Own inversion"</span>)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(theta_linreg)
|
|
sgdreg = SGDRegressor(n_iter = <span style="color: #B452CD">50</span>, penalty=<span style="color: #658b00">None</span>, eta0=<span style="color: #B452CD">0.1</span>)
|
|
sgdreg.fit(x,y.ravel())
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"sgdreg from scikit"</span>)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(sgdreg.intercept_, sgdreg.coef_)
|
|
|
|
|
|
theta = np.random.randn(<span style="color: #B452CD">2</span>,<span style="color: #B452CD">1</span>)
|
|
|
|
eta = <span style="color: #B452CD">0.1</span>
|
|
Niterations = <span style="color: #B452CD">1000</span>
|
|
m = <span style="color: #B452CD">100</span>
|
|
|
|
<span style="color: #8B008B; font-weight: bold">for</span> <span style="color: #658b00">iter</span> <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(Niterations):
|
|
gradients = <span style="color: #B452CD">2.0</span>/m*xb.T.dot(xb.dot(theta)-y)
|
|
theta -= eta*gradients
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"theta frm own gd"</span>)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(theta)
|
|
|
|
xnew = np.array([[<span style="color: #B452CD">0</span>],[<span style="color: #B452CD">2</span>]])
|
|
xbnew = np.c_[np.ones((<span style="color: #B452CD">2</span>,<span style="color: #B452CD">1</span>)), xnew]
|
|
ypredict = xbnew.dot(theta)
|
|
ypredict2 = xbnew.dot(theta_linreg)
|
|
|
|
|
|
n_epochs = <span style="color: #B452CD">50</span>
|
|
t0, t1 = <span style="color: #B452CD">5</span>, <span style="color: #B452CD">50</span>
|
|
m = <span style="color: #B452CD">100</span>
|
|
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">learning_schedule</span>(t):
|
|
<span style="color: #8B008B; font-weight: bold">return</span> t0/(t+t1)
|
|
|
|
theta = np.random.randn(<span style="color: #B452CD">2</span>,<span style="color: #B452CD">1</span>)
|
|
|
|
<span style="color: #8B008B; font-weight: bold">for</span> epoch <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(n_epochs):
|
|
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(m):
|
|
random_index = np.random.randint(m)
|
|
xi = xb[random_index:random_index+<span style="color: #B452CD">1</span>]
|
|
yi = y[random_index:random_index+<span style="color: #B452CD">1</span>]
|
|
gradients = <span style="color: #B452CD">2</span> * xi.T.dot(xi.dot(theta)-yi)
|
|
eta = learning_schedule(epoch*m+i)
|
|
theta = theta - eta*gradients
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(<span style="color: #CD5555">"theta from own sdg"</span>)
|
|
<span style="color: #8B008B; font-weight: bold">print</span>(theta)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
plt.plot(xnew, ypredict, <span style="color: #CD5555">"r-"</span>)
|
|
plt.plot(xnew, ypredict2, <span style="color: #CD5555">"b-"</span>)
|
|
plt.plot(x, y ,<span style="color: #CD5555">'ro'</span>)
|
|
plt.axis([<span style="color: #B452CD">0</span>,<span style="color: #B452CD">2.0</span>,<span style="color: #B452CD">0</span>, <span style="color: #B452CD">15.0</span>])
|
|
plt.xlabel(<span style="color: #CD5555">r'$x$'</span>)
|
|
plt.ylabel(<span style="color: #CD5555">r'$y$'</span>)
|
|
plt.title(<span style="color: #CD5555">r'Random numbers '</span>)
|
|
plt.show()
|
|
</pre></div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec53">Momentum based methods </h2>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec54">Conjugate gradient method </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
In the CG method we define so-called conjugate directions and two vectors
|
|
\( \hat{s} \) and \( \hat{t} \)
|
|
are said to be
|
|
conjugate if
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{s}^T\hat{A}\hat{t}= 0.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
The philosophy of the CG method is to perform searches in various conjugate directions
|
|
of our vectors \( \hat{x}_i \) obeying the above criterion, namely
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}_i^T\hat{A}\hat{x}_j= 0.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
Two vectors are conjugate if they are orthogonal with respect to
|
|
this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \).
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec55">Conjugate gradient method </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
An example is given by the eigenvectors of the matrix
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j,
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
which is zero unless \( i=j \).
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec56">Conjugate gradient method </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size
|
|
\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions.
|
|
Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution
|
|
$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely
|
|
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec57">Conjugate gradient method </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
The coefficients are given by
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
Multiplying with \( \hat{p}_k^T \) from the left gives
|
|
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b},
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
and we can define the coefficients \( \alpha_k \) as
|
|
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k}
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec58">Conjugate gradient method and iterations </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
If we choose the conjugate vectors \( \hat{p}_k \) carefully,
|
|
then we may not need all of them to obtain a good approximation to the solution
|
|
\( \hat{x} \).
|
|
We want to regard the conjugate gradient method as an iterative method.
|
|
This will us to solve systems where \( n \) is so large that the direct
|
|
method would take too much time.
|
|
|
|
<p>
|
|
We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \).
|
|
We can assume without loss of generality that
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}_0=0,
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
or consider the system
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0,
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
instead.
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec59">Conjugate gradient method </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
This suggests taking the first basis vector \( \hat{p}_1 \)
|
|
to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \),
|
|
which equals
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{A}\hat{x}_0-\hat{b},
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
and
|
|
\( \hat{x}_0=0 \) it is equal \( -\hat{b} \).
|
|
The other vectors in the basis will be conjugate to the gradient,
|
|
hence the name conjugate gradient method.
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec60">Conjugate gradient method </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
Let \( \hat{r}_k \) be the residual at the \( k \)-th step:
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
Note that \( \hat{r}_k \) is the negative gradient of \( f \) at
|
|
\( \hat{x}=\hat{x}_k \),
|
|
so the gradient descent method would be to move in the direction \( \hat{r}_k \).
|
|
Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other,
|
|
so we take the direction closest to the gradient \( \hat{r}_k \)
|
|
under the conjugacy constraint.
|
|
This gives the following expression
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k.
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec61">Conjugate gradient method </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
We can also compute the residual iteratively as
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1},
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
which equals
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k),
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
or
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k,
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
|
|
which gives
|
|
|
|
<p> <br>
|
|
$$
|
|
\begin{equation*}
|
|
\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k},
|
|
\end{equation*}
|
|
$$
|
|
<p> <br>
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec62">Simple implementation of the Conjugate gradient algorithm </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
|
|
<!-- code=c++ (!bc cppcod) typeset with pygments style "perldoc" -->
|
|
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%"><span></span> Vector <span style="color: #008b45">ConjugateGradient</span>(Matrix A, Vector b, Vector x0){
|
|
<span style="color: #00688B; font-weight: bold">int</span> dim = x0.Dimension();
|
|
<span style="color: #8B008B; font-weight: bold">const</span> <span style="color: #00688B; font-weight: bold">double</span> tolerance = <span style="color: #B452CD">1.0e-14</span>;
|
|
Vector x(dim),r(dim),v(dim),z(dim);
|
|
<span style="color: #00688B; font-weight: bold">double</span> c,t,d;
|
|
|
|
x = x0;
|
|
r = b - A*x;
|
|
v = r;
|
|
c = dot(r,r);
|
|
<span style="color: #00688B; font-weight: bold">int</span> i = <span style="color: #B452CD">0</span>; IterMax = dim;
|
|
<span style="color: #8B008B; font-weight: bold">while</span>(i <= IterMax){
|
|
z = A*v;
|
|
t = c/dot(v,z);
|
|
x = x + t*v;
|
|
r = r - t*z;
|
|
d = dot(r,r);
|
|
<span style="color: #8B008B; font-weight: bold">if</span>(sqrt(d) < tolerance)
|
|
<span style="color: #8B008B; font-weight: bold">break</span>;
|
|
v = r + (d/c)*v;
|
|
c = d; i++;
|
|
}
|
|
<span style="color: #8B008B; font-weight: bold">return</span> x;
|
|
}
|
|
</pre></div>
|
|
|
|
</div>
|
|
</section>
|
|
|
|
|
|
<section>
|
|
<h2 id="___sec63">Broyden–Fletcher–Goldfarb–Shanno algorithm </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take.
|
|
|
|
<p>
|
|
The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage.
|
|
|
|
<p>
|
|
The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation
|
|
<p> <br>
|
|
$$
|
|
B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}),
|
|
$$
|
|
<p> <br>
|
|
|
|
<p>
|
|
where \( B_{k} \) is an approximation to the Hessian matrix, which is
|
|
updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \)
|
|
is the gradient of the function
|
|
evaluated at \( x_k \).
|
|
A line search in the direction \( p_k \) is then used to
|
|
find the next point \( x_{k+1} \) by minimising
|
|
<p> <br>
|
|
$$
|
|
f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}),
|
|
$$
|
|
<p> <br>
|
|
|
|
over the scalar \( \alpha > 0 \).
|
|
|
|
|
|
</div>
|
|
</section>
|
|
|
|
|
|
|
|
</div> <!-- class="slides" -->
|
|
</div> <!-- class="reveal" -->
|
|
|
|
<script src="reveal.js/lib/js/head.min.js"></script>
|
|
<script src="reveal.js/js/reveal.js"></script>
|
|
|
|
<script>
|
|
// Full list of configuration options available here:
|
|
// https://github.com/hakimel/reveal.js#configuration
|
|
Reveal.initialize({
|
|
|
|
// Display navigation controls in the bottom right corner
|
|
controls: true,
|
|
|
|
// Display progress bar (below the horiz. slider)
|
|
progress: true,
|
|
|
|
// Display the page number of the current slide
|
|
slideNumber: true,
|
|
|
|
// Push each slide change to the browser history
|
|
history: false,
|
|
|
|
// Enable keyboard shortcuts for navigation
|
|
keyboard: true,
|
|
|
|
// Enable the slide overview mode
|
|
overview: true,
|
|
|
|
// Vertical centering of slides
|
|
//center: true,
|
|
center: false,
|
|
|
|
// Enables touch navigation on devices with touch input
|
|
touch: true,
|
|
|
|
// Loop the presentation
|
|
loop: false,
|
|
|
|
// Change the presentation direction to be RTL
|
|
rtl: false,
|
|
|
|
// Turns fragments on and off globally
|
|
fragments: true,
|
|
|
|
// Flags if the presentation is running in an embedded mode,
|
|
// i.e. contained within a limited portion of the screen
|
|
embedded: false,
|
|
|
|
// Number of milliseconds between automatically proceeding to the
|
|
// next slide, disabled when set to 0, this value can be overwritten
|
|
// by using a data-autoslide attribute on your slides
|
|
autoSlide: 0,
|
|
|
|
// Stop auto-sliding after user input
|
|
autoSlideStoppable: true,
|
|
|
|
// Enable slide navigation via mouse wheel
|
|
mouseWheel: false,
|
|
|
|
// Hides the address bar on mobile devices
|
|
hideAddressBar: true,
|
|
|
|
// Opens links in an iframe preview overlay
|
|
previewLinks: false,
|
|
|
|
// Transition style
|
|
transition: 'default', // default/cube/page/concave/zoom/linear/fade/none
|
|
|
|
// Transition speed
|
|
transitionSpeed: 'default', // default/fast/slow
|
|
|
|
// Transition style for full page slide backgrounds
|
|
backgroundTransition: 'default', // default/none/slide/concave/convex/zoom
|
|
|
|
// Number of slides away from the current that are visible
|
|
viewDistance: 3,
|
|
|
|
// Parallax background image
|
|
//parallaxBackgroundImage: '', // e.g. "'https://s3.amazonaws.com/hakim-static/reveal-js/reveal-parallax-1.jpg'"
|
|
|
|
// Parallax background size
|
|
//parallaxBackgroundSize: '' // CSS syntax, e.g. "2100px 900px"
|
|
|
|
theme: Reveal.getQueryHash().theme, // available themes are in reveal.js/css/theme
|
|
transition: Reveal.getQueryHash().transition || 'default', // default/cube/page/concave/zoom/linear/none
|
|
|
|
});
|
|
|
|
Reveal.initialize({
|
|
dependencies: [
|
|
// Cross-browser shim that fully implements classList - https://github.com/eligrey/classList.js/
|
|
{ src: 'reveal.js/lib/js/classList.js', condition: function() { return !document.body.classList; } },
|
|
|
|
// Interpret Markdown in <section> elements
|
|
{ src: 'reveal.js/plugin/markdown/marked.js', condition: function() { return !!document.querySelector( '[data-markdown]' ); } },
|
|
{ src: 'reveal.js/plugin/markdown/markdown.js', condition: function() { return !!document.querySelector( '[data-markdown]' ); } },
|
|
|
|
// Syntax highlight for <code> elements
|
|
{ src: 'reveal.js/plugin/highlight/highlight.js', async: true, callback: function() { hljs.initHighlightingOnLoad(); } },
|
|
|
|
// Zoom in and out with Alt+click
|
|
{ src: 'reveal.js/plugin/zoom-js/zoom.js', async: true, condition: function() { return !!document.body.classList; } },
|
|
|
|
// Speaker notes
|
|
{ src: 'reveal.js/plugin/notes/notes.js', async: true, condition: function() { return !!document.body.classList; } },
|
|
|
|
// Remote control your reveal.js presentation using a touch device
|
|
//{ src: 'reveal.js/plugin/remotes/remotes.js', async: true, condition: function() { return !!document.body.classList; } },
|
|
|
|
// MathJax
|
|
//{ src: 'reveal.js/plugin/math/math.js', async: true }
|
|
]
|
|
});
|
|
|
|
Reveal.initialize({
|
|
|
|
// The "normal" size of the presentation, aspect ratio will be preserved
|
|
// when the presentation is scaled to fit different resolutions. Can be
|
|
// specified using percentage units.
|
|
width: 1170, // original: 960,
|
|
height: 700,
|
|
|
|
// Factor of the display size that should remain empty around the content
|
|
margin: 0.1,
|
|
|
|
// Bounds for smallest/largest possible scale to apply to content
|
|
minScale: 0.2,
|
|
maxScale: 1.0
|
|
|
|
});
|
|
</script>
|
|
|
|
<!-- begin footer logo
|
|
<div style="position: absolute; bottom: 0px; left: 0; margin-left: 0px">
|
|
<img src="somelogo.png">
|
|
</div>
|
|
end footer logo -->
|
|
|
|
|
|
|
|
</body>
|
|
</html>
|