2171 lines
119 KiB
HTML
2171 lines
119 KiB
HTML
<!--
|
||
Automatically generated HTML file from DocOnce source
|
||
(https://github.com/hplgit/doconce/)
|
||
-->
|
||
<html>
|
||
<head>
|
||
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
||
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
|
||
<meta name="description" content="Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods">
|
||
|
||
<title>Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods</title>
|
||
|
||
|
||
<style type="text/css">
|
||
/* bloodish style */
|
||
|
||
body {
|
||
font-family: Helvetica, Verdana, Arial, Sans-serif;
|
||
color: #404040;
|
||
background: #ffffff;
|
||
}
|
||
h1 { font-size: 1.8em; color: #8A0808; }
|
||
h2 { font-size: 1.6em; color: #8A0808; }
|
||
h3 { font-size: 1.4em; color: #8A0808; }
|
||
h4 { color: #8A0808; }
|
||
a { color: #8A0808; text-decoration:none; }
|
||
tt { font-family: "Courier New", Courier; }
|
||
/* pre style removed because it will interfer with pygments */
|
||
p { text-indent: 0px; }
|
||
hr { border: 0; width: 80%; border-bottom: 1px solid #aaa}
|
||
p.caption { width: 80%; font-style: normal; text-align: left; }
|
||
hr.figure { border: 0; width: 80%; border-bottom: 1px solid #aaa}
|
||
.alert-text-small { font-size: 80%; }
|
||
.alert-text-large { font-size: 130%; }
|
||
.alert-text-normal { font-size: 90%; }
|
||
.alert {
|
||
padding:8px 35px 8px 14px; margin-bottom:18px;
|
||
text-shadow:0 1px 0 rgba(255,255,255,0.5);
|
||
border:1px solid #bababa;
|
||
border-radius: 4px;
|
||
-webkit-border-radius: 4px;
|
||
-moz-border-radius: 4px;
|
||
color: #555;
|
||
background-color: #f8f8f8;
|
||
background-position: 10px 5px;
|
||
background-repeat: no-repeat;
|
||
background-size: 38px;
|
||
padding-left: 55px;
|
||
width: 75%;
|
||
}
|
||
.alert-block {padding-top:14px; padding-bottom:14px}
|
||
.alert-block > p, .alert-block > ul {margin-bottom:1em}
|
||
.alert li {margin-top: 1em}
|
||
.alert-block p+p {margin-top:5px}
|
||
.alert-notice { background-image: url(https://cdn.rawgit.com/hplgit/doconce/master/bundled/html_images/small_gray_notice.png); }
|
||
.alert-summary { background-image:url(https://cdn.rawgit.com/hplgit/doconce/master/bundled/html_images/small_gray_summary.png); }
|
||
.alert-warning { background-image: url(https://cdn.rawgit.com/hplgit/doconce/master/bundled/html_images/small_gray_warning.png); }
|
||
.alert-question {background-image:url(https://cdn.rawgit.com/hplgit/doconce/master/bundled/html_images/small_gray_question.png); }
|
||
|
||
div { text-align: justify; text-justify: inter-word; }
|
||
</style>
|
||
|
||
|
||
</head>
|
||
|
||
<!-- tocinfo
|
||
{'highest level': 2,
|
||
'sections': [('Optimization, the central part of any Machine Learning '
|
||
'algortithm',
|
||
2,
|
||
None,
|
||
'___sec0'),
|
||
('Revisiting our Logistic Regression case', 2, None, '___sec1'),
|
||
('The equations to solve', 2, None, '___sec2'),
|
||
("Solving using Newton-Raphson's method", 2, None, '___sec3'),
|
||
("Brief reminder on Newton-Raphson's method", 2, None, '___sec4'),
|
||
('The equations', 2, None, '___sec5'),
|
||
('Simple geometric interpretation', 2, None, '___sec6'),
|
||
('Extending to more than one variable', 2, None, '___sec7'),
|
||
('Steepest descent', 2, None, '___sec8'),
|
||
('More on Steepest descent', 2, None, '___sec9'),
|
||
('The ideal', 2, None, '___sec10'),
|
||
('The sensitiveness of the gradient descent',
|
||
2,
|
||
None,
|
||
'___sec11'),
|
||
('Convex functions', 2, None, '___sec12'),
|
||
('Convex function', 2, None, '___sec13'),
|
||
('Conditions on convex functions', 2, None, '___sec14'),
|
||
('More on convex functions', 2, None, '___sec15'),
|
||
('Some simple problems', 2, None, '___sec16'),
|
||
('Standard steepest descent', 2, None, '___sec17'),
|
||
('Gradient method', 2, None, '___sec18'),
|
||
('Steepest descent method', 2, None, '___sec19'),
|
||
('Steepest descent method', 2, None, '___sec20'),
|
||
('Final expressions', 2, None, '___sec21'),
|
||
('Simple codes for steepest descent and conjugate gradient '
|
||
'using a $2\\times 2$ matrix, in c++, Python code to come',
|
||
2,
|
||
None,
|
||
'___sec22'),
|
||
('The routine for the steepest descent method',
|
||
2,
|
||
None,
|
||
'___sec23'),
|
||
('Steepest descent example', 2, None, '___sec24'),
|
||
('Conjugate gradient', 2, None, '___sec25'),
|
||
('Revisiting our first homework', 2, None, '___sec26'),
|
||
('Gradient descent example', 2, None, '___sec27'),
|
||
('The derivative of the cost/loss function', 2, None, '___sec28'),
|
||
('The Hessian matrix', 2, None, '___sec29'),
|
||
('Simple program', 2, None, '___sec30'),
|
||
('Gradient Descent Example', 2, None, '___sec31'),
|
||
('And a corresponding example using _scikit-learn_',
|
||
2,
|
||
None,
|
||
'___sec32'),
|
||
('Gradient descent and Ridge', 2, None, '___sec33'),
|
||
('Automatic differentiation', 2, None, '___sec34'),
|
||
('Using autograd', 2, None, '___sec35'),
|
||
('Autograd with more complicated functions', 2, None, '___sec36'),
|
||
('More complicated functions using the elements of their '
|
||
'arguments directly',
|
||
2,
|
||
None,
|
||
'___sec37'),
|
||
('Functions using mathematical functions from Numpy',
|
||
2,
|
||
None,
|
||
'___sec38'),
|
||
('More autograd', 2, None, '___sec39'),
|
||
('And with loops', 2, None, '___sec40'),
|
||
('Using recursion', 2, None, '___sec41'),
|
||
('Unsupported functions', 2, None, '___sec42'),
|
||
('The syntax a.dot(b) when finding the dot product',
|
||
2,
|
||
None,
|
||
'___sec43'),
|
||
('Recommended to avoid', 2, None, '___sec44'),
|
||
('Stochastic Gradient Descent', 2, None, '___sec45'),
|
||
('Computation of gradients', 2, None, '___sec46'),
|
||
('SGD example', 2, None, '___sec47'),
|
||
('The gradient step', 2, None, '___sec48'),
|
||
('Simple example code', 2, None, '___sec49'),
|
||
('When do we stop?', 2, None, '___sec50'),
|
||
('Slightly different approach', 2, None, '___sec51'),
|
||
('Program for stochastic gradient', 2, None, '___sec52'),
|
||
('Momentum based methods', 2, None, '___sec53'),
|
||
('Conjugate gradient method', 2, None, '___sec54'),
|
||
('Conjugate gradient method', 2, None, '___sec55'),
|
||
('Conjugate gradient method', 2, None, '___sec56'),
|
||
('Conjugate gradient method', 2, None, '___sec57'),
|
||
('Conjugate gradient method and iterations', 2, None, '___sec58'),
|
||
('Conjugate gradient method', 2, None, '___sec59'),
|
||
('Conjugate gradient method', 2, None, '___sec60'),
|
||
('Conjugate gradient method', 2, None, '___sec61'),
|
||
('Simple implementation of the Conjugate gradient algorithm',
|
||
2,
|
||
None,
|
||
'___sec62'),
|
||
('Broyden–Fletcher–Goldfarb–Shanno algorithm',
|
||
2,
|
||
None,
|
||
'___sec63')]}
|
||
end of tocinfo -->
|
||
|
||
<body>
|
||
|
||
|
||
|
||
<script type="text/x-mathjax-config">
|
||
MathJax.Hub.Config({
|
||
TeX: {
|
||
equationNumbers: { autoNumber: "AMS" },
|
||
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
|
||
}
|
||
});
|
||
</script>
|
||
<script type="text/javascript" async
|
||
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
|
||
</script>
|
||
|
||
|
||
|
||
|
||
<!-- ------------------- main content ---------------------- -->
|
||
|
||
|
||
|
||
<center><h1>Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods</h1></center> <!-- document title -->
|
||
|
||
<p>
|
||
<!-- author(s): Morten Hjorth-Jensen -->
|
||
|
||
<center>
|
||
<b>Morten Hjorth-Jensen</b> [1, 2]
|
||
</center>
|
||
|
||
<p>
|
||
<!-- institution(s) -->
|
||
|
||
<center>[1] <b>Department of Physics, University of Oslo</b></center>
|
||
<center>[2] <b>Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University</b></center>
|
||
<br>
|
||
<p>
|
||
<center><h4>Oct 6, 2018</h4></center> <!-- date -->
|
||
<br>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec0">Optimization, the central part of any Machine Learning algortithm </h2>
|
||
|
||
<p>
|
||
Almost every problem in machine learning and data science starts with
|
||
a dataset \( X \), a model \( g(\beta) \), which is a function of the
|
||
parameters \( \beta \) and a cost function \( C(X, g(\beta)) \) that allows
|
||
us to judge how well the model \( g(\beta) \) explains the observations
|
||
\( X \). The model is fit by finding the values of \( \beta \) that minimize
|
||
the cost function. Ideally we would be able to solve for \( \beta \)
|
||
analytically, however this is not possible in general and we must use
|
||
some approximative/numerical method to compute the minimum.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec1">Revisiting our Logistic Regression case </h2>
|
||
|
||
<p>
|
||
In our discussion on Logistic Regression we studied the
|
||
case of
|
||
two classes, with \( y_i \) either
|
||
\( 0 \) or \( 1 \). Furthermore we assumed also that we have only two
|
||
parameters \( \beta \) in our fitting, that is we
|
||
defined probabilities
|
||
|
||
$$
|
||
\begin{align*}
|
||
p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\
|
||
p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}),
|
||
\end{align*}
|
||
$$
|
||
|
||
where \( \hat{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec2">The equations to solve </h2>
|
||
|
||
<p>
|
||
Our compact equations used a definition of a vector \( \hat{y} \) with \( n \)
|
||
elements \( y_i \), an \( n\times p \) matrix \( \hat{X} \) which contains the
|
||
\( x_i \) values and a vector \( \hat{p} \) of fitted probabilities
|
||
\( p(y_i\vert x_i,\hat{\beta}) \). We rewrote in a more compact form
|
||
the first derivative of the cost function as
|
||
|
||
$$
|
||
\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right).
|
||
$$
|
||
|
||
<p>
|
||
If we in addition define a diagonal matrix \( \hat{W} \) with elements
|
||
\( p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta}) \), we can obtain a compact expression of the second derivative as
|
||
|
||
$$
|
||
\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}.
|
||
$$
|
||
|
||
This defines what is called the Hessian matrix.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec3">Solving using Newton-Raphson's method </h2>
|
||
|
||
<p>
|
||
If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives.
|
||
|
||
<p>
|
||
Our iterative scheme is then given by
|
||
|
||
$$
|
||
\hat{\beta}^{\mathrm{new}} = \hat{\beta}^{\mathrm{old}}-\left(\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T}\right)^{-1}_{\hat{\beta}^{\mathrm{old}}}\times \left(\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}}\right)_{\hat{\beta}^{\mathrm{old}}},
|
||
$$
|
||
|
||
or in matrix form as
|
||
|
||
$$
|
||
\hat{\beta}^{\mathrm{new}} = \hat{\beta}^{\mathrm{old}}-\left(\hat{X}^T\hat{W}\hat{X} \right)^{-1}\times \left(-\hat{X}^T(\hat{y}-\hat{p}) \right)_{\hat{\beta}^{\mathrm{old}}}.
|
||
$$
|
||
|
||
The right-hand side is computed with the old values of \( \beta \).
|
||
|
||
<p>
|
||
If we can compute these matrices, in particular the Hessian, the above is often the easiest method to implement.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec4">Brief reminder on Newton-Raphson's method </h2>
|
||
|
||
<p>
|
||
Let us quickly remind ourselves how we derive the above method.
|
||
|
||
<p>
|
||
Perhaps the most celebrated of all one-dimensional root-finding
|
||
routines is Newton's method, also called the Newton-Raphson
|
||
method. This method requires the evaluation of both the
|
||
function \( f \) and its derivative \( f' \) at arbitrary points.
|
||
If you can only calculate the derivative
|
||
numerically and/or your function is not of the smooth type, we
|
||
normally discourage the use of this method.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec5">The equations </h2>
|
||
|
||
<p>
|
||
The Newton-Raphson formula consists geometrically of extending the
|
||
tangent line at a current point until it crosses zero, then setting
|
||
the next guess to the abscissa of that zero-crossing. The mathematics
|
||
behind this method is rather simple. Employing a Taylor expansion for
|
||
\( x \) sufficiently close to the solution \( s \), we have
|
||
|
||
$$
|
||
f(s)=0=f(x)+(s-x)f'(x)+\frac{(s-x)^2}{2}f''(x) +\dots.
|
||
\label{eq:taylornr}
|
||
$$
|
||
|
||
<p>
|
||
For small enough values of the function and for well-behaved
|
||
functions, the terms beyond linear are unimportant, hence we obtain
|
||
|
||
$$
|
||
f(x)+(s-x)f'(x)\approx 0,
|
||
$$
|
||
|
||
yielding
|
||
$$
|
||
s\approx x-\frac{f(x)}{f'(x)}.
|
||
$$
|
||
|
||
<p>
|
||
Having in mind an iterative procedure, it is natural to start iterating with
|
||
$$
|
||
x_{n+1}=x_n-\frac{f(x_n)}{f'(x_n)}.
|
||
$$
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec6">Simple geometric interpretation </h2>
|
||
|
||
<p>
|
||
The above is Newton-Raphson's method. It has a simple geometric
|
||
interpretation, namely \( x_{n+1} \) is the point where the tangent from
|
||
\( (x_n,f(x_n)) \) crosses the \( x \)-axis. Close to the solution,
|
||
Newton-Raphson converges fast to the desired result. However, if we
|
||
are far from a root, where the higher-order terms in the series are
|
||
important, the Newton-Raphson formula can give grossly inaccurate
|
||
results. For instance, the initial guess for the root might be so far
|
||
from the true root as to let the search interval include a local
|
||
maximum or minimum of the function. If an iteration places a trial
|
||
guess near such a local extremum, so that the first derivative nearly
|
||
vanishes, then Newton-Raphson may fail totally
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec7">Extending to more than one variable </h2>
|
||
|
||
<p>
|
||
Newton's method can be generalized to systems of several non-linear equations
|
||
and variables. Consider the case with two equations
|
||
$$
|
||
\begin{array}{cc} f_1(x_1,x_2) &=0\\
|
||
f_2(x_1,x_2) &=0,\end{array}
|
||
$$
|
||
|
||
which we Taylor expand to obtain
|
||
|
||
$$
|
||
\begin{array}{cc} 0=f_1(x_1+h_1,x_2+h_2)=&f_1(x_1,x_2)+h_1
|
||
\partial f_1/\partial x_1+h_2
|
||
\partial f_1/\partial x_2+\dots\\
|
||
0=f_2(x_1+h_1,x_2+h_2)=&f_2(x_1,x_2)+h_1
|
||
\partial f_2/\partial x_1+h_2
|
||
\partial f_2/\partial x_2+\dots
|
||
\end{array}.
|
||
$$
|
||
|
||
Defining the Jacobian matrix \( {\bf \hat{J}} \) we have
|
||
$$
|
||
{\bf \hat{J}}=\left( \begin{array}{cc}
|
||
\partial f_1/\partial x_1 & \partial f_1/\partial x_2 \\
|
||
\partial f_2/\partial x_1 &\partial f_2/\partial x_2
|
||
\end{array} \right),
|
||
$$
|
||
|
||
we can rephrase Newton's method as
|
||
$$
|
||
\left(\begin{array}{c} x_1^{n+1} \\ x_2^{n+1} \end{array} \right)=
|
||
\left(\begin{array}{c} x_1^{n} \\ x_2^{n} \end{array} \right)+
|
||
\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right),
|
||
$$
|
||
|
||
where we have defined
|
||
$$
|
||
\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right)=
|
||
-{\bf \hat{J}}^{-1}
|
||
\left(\begin{array}{c} f_1(x_1^{n},x_2^{n}) \\ f_2(x_1^{n},x_2^{n}) \end{array} \right).
|
||
$$
|
||
|
||
We need thus to compute the inverse of the Jacobian matrix and it
|
||
is to understand that difficulties may
|
||
arise in case \( {\bf \hat{J}} \) is nearly singular.
|
||
|
||
<p>
|
||
It is rather straightforward to extend the above scheme to systems of
|
||
more than two non-linear equations. In our case, the Jacobian matrix is given by the Hessian that represents the second derivative of cost function.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec8">Steepest descent </h2>
|
||
|
||
<p>
|
||
The basic idea of gradient descent is
|
||
that a function \( F(\mathbf{x}) \),
|
||
\( \mathbf{x} \equiv (x_1,\cdots,x_n) \), decreases fastest if one goes from \( \bf {x} \) in the
|
||
direction of the negative gradient \( -\nabla F(\mathbf{x}) \).
|
||
|
||
<p>
|
||
It can be shown that if
|
||
$$
|
||
\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k),
|
||
$$
|
||
|
||
with \( \gamma_k > 0 \).
|
||
|
||
<p>
|
||
For \( \gamma_k \) small enough, then \( F(\mathbf{x}_{k+1}) \leq
|
||
F(\mathbf{x}_k) \). This means that for a sufficiently small \( \gamma_k \)
|
||
we are always moving towards smaller function values, i.e a minimum.
|
||
|
||
<p>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec9">More on Steepest descent </h2>
|
||
|
||
<p>
|
||
The previous observation is the basis of the method of steepest
|
||
descent, which is also referred to as just gradient descent (GD). One
|
||
starts with an initial guess \( \mathbf{x}_0 \) for a minimum of \( F \) and
|
||
computes new approximations according to
|
||
|
||
$$
|
||
\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), \ \ k \geq 0.
|
||
$$
|
||
|
||
<p>
|
||
The parameter \( \gamma_k \) is often referred to as the step length or
|
||
the learning rate within the context of Machine Learning.
|
||
|
||
<p>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec10">The ideal </h2>
|
||
|
||
<p>
|
||
Ideally the sequence \( \{\mathbf{x}_k \}_{k=0} \) converges to a global
|
||
minimum of the function \( F \). In general we do not know if we are in a
|
||
global or local minimum. In the special case when \( F \) is a convex
|
||
function, all local minima are also global minima, so in this case
|
||
gradient descent can converge to the global solution. The advantage of
|
||
this scheme is that it is conceptually simple and straightforward to
|
||
implement. However the method in this form has some severe
|
||
limitations:
|
||
|
||
<p>
|
||
In machine learing we are often faced with non-convex high dimensional
|
||
cost functions with many local minima. Since GD is deterministic we
|
||
will get stuck in a local minimum, if the method converges, unless we
|
||
have a very good intial guess. This also implies that the scheme is
|
||
sensitive to the chosen initial condition.
|
||
|
||
<p>
|
||
Note that the gradient is a function of \( \mathbf{x} =
|
||
(x_1,\cdots,x_n) \) which makes it expensive to compute numerically.
|
||
|
||
<p>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec11">The sensitiveness of the gradient descent </h2>
|
||
|
||
<p>
|
||
The gradient descent method
|
||
is sensitive to the choice of learning rate \( \gamma_k \). This is due
|
||
to the fact that we are only guaranteed that \( F(\mathbf{x}_{k+1}) \leq
|
||
F(\mathbf{x}_k) \) for sufficiently small \( \gamma_k \). The problem is to
|
||
determine an optimal learning rate. If the learning rate is chosen too
|
||
small the method will take a long time to converge and if it is too
|
||
large we can experience erratic behavior.
|
||
|
||
<p>
|
||
Many of these shortcomings can be alleviated by introducing
|
||
randomness. One such method is that of Stochastic Gradient Descent
|
||
(SGD), see below.
|
||
|
||
<p>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec12">Convex functions </h2>
|
||
|
||
<p>
|
||
Ideally we want our cost/loss function to be convex(concave).
|
||
|
||
<p>
|
||
First we give the definition of a convex set: A set \( C \) in
|
||
\( \mathbb{R}^n \) is said to be convex if, for all \( x \) and \( y \) in \( C \) and
|
||
all \( t \in (0,1) \) , the point \( (1 − t)x + ty \) also belongs to
|
||
C. Geometrically this means that every point on the line segment
|
||
connecting \( x \) and \( y \) is in \( C \) as discussed below.
|
||
|
||
<p>
|
||
The convex subsets of \( \mathbb{R} \) are the intervals of
|
||
\( \mathbb{R} \). Examples of convex sets of \( \mathbb{R}^2 \) are the
|
||
regular polygons (triangles, rectangles, pentagons, etc...).
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec13">Convex function </h2>
|
||
|
||
<p>
|
||
<b>Convex function</b>: Let \( X \subset \mathbb{R}^n \) be a convex set. Assume that the function \( f: X \rightarrow \mathbb{R} \) is continuous, then \( f \) is said to be convex if $$f(tx_1 + (1-t)x_2) \leq tf(x_1) + (1-t)f(x_2) $$ for all \( x_1, x_2 \in X \) and for all \( t \in [0,1] \). If \( \leq \) is replaced with a strict inequaltiy in the definition, we demand \( x_1 \neq x_2 \) and \( t\in(0,1) \) then \( f \) is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting \( f(x_1) \) and \( f(x_2) \), the value of the function on the interval \( [x_1,x_2] \) is always below the line as illustrated below.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec14">Conditions on convex functions </h2>
|
||
|
||
<p>
|
||
In the following we state first and second-order conditions which
|
||
ensures convexity of a function \( f \). We write \( D_f \) to denote the
|
||
domain of \( f \), i.e the subset of \( R^n \) where \( f \) is defined. For more
|
||
details and proofs we refer to: <a href="http://stanford.edu/boyd/cvxbook/, 2004" target="_blank">S. Boyd and L. Vandenberghe. Convex Optimization. Cambridge University Press</a>.
|
||
|
||
<p>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b>First order condition.</b>
|
||
<p>
|
||
Suppose \( f \) is differentiable (i.e \( \nabla f(x) \) is well defined for
|
||
all \( x \) in the domain of \( f \)). Then \( f \) is convex if and only if \( D_f \)
|
||
is a convex set and $$f(y) \geq f(x) + \nabla f(x)^T (y-x) $$ holds
|
||
for all \( x,y \in D_f \). This condition means that for a convex function
|
||
the first order Taylor expansion (right hand side above) at any point
|
||
a global under estimator of the function. To convince yourself you can
|
||
make a drawing of \( f(x) = x^2+1 \) and draw the tangent line to \( f(x) \) and
|
||
note that it is always below the graph.
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b>Second order condition.</b>
|
||
<p>
|
||
Assume that \( f \) is twice
|
||
differentiable, i.e the Hessian matrix exists at each point in
|
||
\( D_f \). Then \( f \) is convex if and only if \( D_f \) is a convex set and its
|
||
Hessian is positive semi-definite for all \( x\in D_f \). For a
|
||
single-variable function this reduces to \( f''(x) \geq 0 \). Geometrically this means that \( f \) has nonnegative curvature
|
||
everywhere.
|
||
</div>
|
||
|
||
|
||
<p>
|
||
This condition is particularly useful since it gives us an procedure for determining if the function under consideration is convex, apart from using the definition.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec15">More on convex functions </h2>
|
||
|
||
<p>
|
||
The next result is of great importance to us and the reason why we are
|
||
going on about convex functions. In machine learning we frequently
|
||
have to minimize a loss/cost function in order to find the best
|
||
parameters for the model we are considering.
|
||
|
||
<p>
|
||
Ideally we want the
|
||
global minimum (for high-dimensional models it is hard to know
|
||
if we have local or global minimum). However, if the cost/loss function
|
||
is convex the following result provides invaluable information:
|
||
|
||
<p>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b>Any minimum is global for convex functions.</b>
|
||
<p>
|
||
Consider the problem of finding \( x \in \mathbb{R}^n \) such that \( f(x) \)
|
||
is minimal, where \( f \) is convex and differentiable. Then, any point
|
||
\( x^* \) that satisfies \( \nabla f(x^*) = 0 \) is a global minimum.
|
||
</div>
|
||
|
||
|
||
<p>
|
||
This result means that if we know that the cost/loss function is convex and we are able to find a minimum, we are guaranteed that it is a global minimum.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec16">Some simple problems </h2>
|
||
|
||
<ol>
|
||
<li> Show that \( f(x)=x^2 \) is convex for \( x \in \mathbb{R} \) using the definition of convexity. Hint: If you re-write the definition, \( f \) is convex if the following holds for all \( x,y \in D_f \) and any \( \lambda \in [0,1] \) $\lambda f(x)+(1-\lambda)f(y)-f(\lambda x + (1-\lambda) y ) \geq 0$.</li>
|
||
<li> Using the second order condition show that the following functions are convex on the specified domain.</li>
|
||
|
||
<ul>
|
||
<li> \( f(x) = e^x \) is convex for \( x \in \mathbb{R} \).</li>
|
||
<li> \( g(x) = -\ln(x) \) is convex for \( x \in (0,\infty) \).</li>
|
||
</ul>
|
||
|
||
<li> Let \( f(x) = x^2 \) and \( g(x) = e^x \). Show that \( f(g(x)) \) and \( g(f(x)) \) is convex for \( x \in \mathbb{R} \). Also show that if \( f(x) \) is any convex function than \( h(x) = e^{f(x)} \) is convex.</li>
|
||
<li> A norm is any function that satisfy the following properties</li>
|
||
|
||
<ul>
|
||
<li> \( f(\alpha x) = |\alpha| f(x) \) for all \( \alpha \in \mathbb{R} \).</li>
|
||
<li> \( f(x+y) \leq f(x) + f(y) \)</li>
|
||
<li> \( f(x) \leq 0 \) for all \( x \in \mathbb{R}^n \) with equality if and only if \( x = 0 \)</li>
|
||
</ul>
|
||
|
||
</ol>
|
||
|
||
Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this).
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec17">Standard steepest descent </h2>
|
||
|
||
<p>
|
||
Before we proceed, we would like to discuss the approach called the
|
||
<b>standard Steepest descent</b>, which again leads to us having to be able
|
||
to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG).
|
||
|
||
<p>
|
||
<a href="https://www.cs.cmu.edu/~quake-papers/painless-conjugate-gradient.pdf" target="_blank">The success of the CG method</a>
|
||
for finding solutions of non-linear problems is based on the theory
|
||
of conjugate gradients for linear systems of equations. It belongs to
|
||
the class of iterative methods for solving problems from linear
|
||
algebra of the type
|
||
$$
|
||
\begin{equation*}
|
||
\hat{A}\hat{x} = \hat{b}.
|
||
\end{equation*}
|
||
$$
|
||
|
||
<p>
|
||
In the iterative process we end up with a problem like
|
||
|
||
$$
|
||
\begin{equation*}
|
||
\hat{r}= \hat{b}-\hat{A}\hat{x},
|
||
\end{equation*}
|
||
$$
|
||
|
||
where \( \hat{r} \) is the so-called residual or error in the iterative process.
|
||
|
||
<p>
|
||
When we have found the exact solution, \( \hat{r}=0 \).
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec18">Gradient method </h2>
|
||
|
||
<p>
|
||
The residual is zero when we reach the minimum of the quadratic equation
|
||
$$
|
||
\begin{equation*}
|
||
P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b},
|
||
\end{equation*}
|
||
$$
|
||
|
||
<p>
|
||
with the constraint that the matrix \( \hat{A} \) is positive definite and
|
||
symmetric. This defines also the Hessian and we want it to be positive definite.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec19">Steepest descent method </h2>
|
||
|
||
<p>
|
||
We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \).
|
||
We can assume without loss of generality that
|
||
$$
|
||
\begin{equation*}
|
||
\hat{x}_0=0,
|
||
\end{equation*}
|
||
$$
|
||
|
||
or consider the system
|
||
$$
|
||
\begin{equation*}
|
||
\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0,
|
||
\end{equation*}
|
||
$$
|
||
|
||
instead.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec20">Steepest descent method </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form
|
||
$$
|
||
\begin{equation*}
|
||
f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n.
|
||
\end{equation*}
|
||
$$
|
||
|
||
This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition)
|
||
to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \),
|
||
which equals
|
||
$$
|
||
\begin{equation*}
|
||
\hat{A}\hat{x}_0-\hat{b},
|
||
\end{equation*}
|
||
$$
|
||
|
||
and
|
||
\( \hat{x}_0=0 \) it is equal \( -\hat{b} \).
|
||
|
||
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec21">Final expressions </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
We can compute the residual iteratively as
|
||
$$
|
||
\begin{equation*}
|
||
\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1},
|
||
\end{equation*}
|
||
$$
|
||
|
||
which equals
|
||
$$
|
||
\begin{equation*}
|
||
\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k),
|
||
\end{equation*}
|
||
$$
|
||
|
||
or
|
||
$$
|
||
\begin{equation*}
|
||
(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k,
|
||
\end{equation*}
|
||
$$
|
||
|
||
which gives
|
||
|
||
$$
|
||
\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k}
|
||
$$
|
||
|
||
leading to the iterative scheme
|
||
$$
|
||
\begin{equation*}
|
||
\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k},
|
||
\end{equation*}
|
||
$$
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec22">Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
<p>
|
||
|
||
<!-- code=c++ (!bc cppcod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #BC7A00">#include</span> <span style="color: #408080; font-style: italic"><cmath></span><span style="color: #BC7A00"></span>
|
||
<span style="color: #BC7A00">#include</span> <span style="color: #408080; font-style: italic"><iostream></span><span style="color: #BC7A00"></span>
|
||
<span style="color: #BC7A00">#include</span> <span style="color: #408080; font-style: italic"><fstream></span><span style="color: #BC7A00"></span>
|
||
<span style="color: #BC7A00">#include</span> <span style="color: #408080; font-style: italic"><iomanip></span><span style="color: #BC7A00"></span>
|
||
<span style="color: #BC7A00">#include</span> <span style="color: #408080; font-style: italic">"vectormatrixclass.h"</span><span style="color: #BC7A00"></span>
|
||
<span style="color: #008000; font-weight: bold">using</span> <span style="color: #008000; font-weight: bold">namespace</span> std;
|
||
<span style="color: #408080; font-style: italic">// Main function begins here</span>
|
||
<span style="color: #B00040">int</span> <span style="color: #0000FF">main</span>(<span style="color: #B00040">int</span> argc, <span style="color: #B00040">char</span> <span style="color: #666666">*</span> argv[]){
|
||
<span style="color: #B00040">int</span> dim <span style="color: #666666">=</span> <span style="color: #666666">2</span>;
|
||
Vector x(dim),xsd(dim), b(dim),x0(dim);
|
||
Matrix A(dim,dim);
|
||
|
||
<span style="color: #408080; font-style: italic">// Set our initial guess</span>
|
||
x0(<span style="color: #666666">0</span>) <span style="color: #666666">=</span> x0(<span style="color: #666666">1</span>) <span style="color: #666666">=</span> <span style="color: #666666">0</span>;
|
||
<span style="color: #408080; font-style: italic">// Set the matrix</span>
|
||
A(<span style="color: #666666">0</span>,<span style="color: #666666">0</span>) <span style="color: #666666">=</span> <span style="color: #666666">3</span>; A(<span style="color: #666666">1</span>,<span style="color: #666666">0</span>) <span style="color: #666666">=</span> <span style="color: #666666">2</span>; A(<span style="color: #666666">0</span>,<span style="color: #666666">1</span>) <span style="color: #666666">=</span> <span style="color: #666666">2</span>; A(<span style="color: #666666">1</span>,<span style="color: #666666">1</span>) <span style="color: #666666">=</span> <span style="color: #666666">6</span>;
|
||
b(<span style="color: #666666">0</span>) <span style="color: #666666">=</span> <span style="color: #666666">2</span>; b(<span style="color: #666666">1</span>) <span style="color: #666666">=</span> <span style="color: #666666">-8</span>;
|
||
cout <span style="color: #666666"><<</span> <span style="color: #BA2121">"The Matrix A that we are using: "</span> <span style="color: #666666"><<</span> endl;
|
||
A.Print();
|
||
cout <span style="color: #666666"><<</span> endl;
|
||
xsd <span style="color: #666666">=</span> SteepestDescent(A,b,x0);
|
||
cout <span style="color: #666666"><<</span> <span style="color: #BA2121">"The approximate solution using Steepest Descent is: "</span> <span style="color: #666666"><<</span> endl;
|
||
xsd.Print();
|
||
cout <span style="color: #666666"><<</span> endl;
|
||
}
|
||
</pre></div>
|
||
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec23">The routine for the steepest descent method </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
<p>
|
||
|
||
<!-- code=c++ (!bc cppcod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>Vector <span style="color: #0000FF">SteepestDescent</span>(Matrix A, Vector b, Vector x0){
|
||
<span style="color: #B00040">int</span> IterMax, i;
|
||
<span style="color: #B00040">int</span> dim <span style="color: #666666">=</span> x0.Dimension();
|
||
<span style="color: #008000; font-weight: bold">const</span> <span style="color: #B00040">double</span> tolerance <span style="color: #666666">=</span> <span style="color: #666666">1.0e-14</span>;
|
||
Vector x(dim),f(dim),z(dim);
|
||
<span style="color: #B00040">double</span> c,alpha,d;
|
||
IterMax <span style="color: #666666">=</span> <span style="color: #666666">30</span>;
|
||
x <span style="color: #666666">=</span> x0;
|
||
r <span style="color: #666666">=</span> A<span style="color: #666666">*</span>x<span style="color: #666666">-</span>b;
|
||
i <span style="color: #666666">=</span> <span style="color: #666666">0</span>;
|
||
<span style="color: #008000; font-weight: bold">while</span> (i <span style="color: #666666"><=</span> IterMax){
|
||
z <span style="color: #666666">=</span> A<span style="color: #666666">*</span>r;
|
||
c <span style="color: #666666">=</span> dot(r,r);
|
||
alpha <span style="color: #666666">=</span> c<span style="color: #666666">/</span>dot(r,z);
|
||
x <span style="color: #666666">=</span> x <span style="color: #666666">-</span> alpha<span style="color: #666666">*</span>r;
|
||
r <span style="color: #666666">=</span> A<span style="color: #666666">*</span>x<span style="color: #666666">-</span>b;
|
||
<span style="color: #008000; font-weight: bold">if</span>(sqrt(dot(r,r)) <span style="color: #666666"><</span> tolerance) <span style="color: #008000; font-weight: bold">break</span>;
|
||
i<span style="color: #666666">++</span>;
|
||
}
|
||
<span style="color: #008000; font-weight: bold">return</span> x;
|
||
}
|
||
</pre></div>
|
||
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec24">Steepest descent example </h2>
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy.linalg</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">la</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">scipy.optimize</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">sopt</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pt</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">mpl_toolkits.mplot3d</span> <span style="color: #008000; font-weight: bold">import</span> axes3d
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f</span>(x):
|
||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">0.5*</span>x[<span style="color: #666666">0</span>]<span style="color: #666666">**2</span> <span style="color: #666666">+</span> <span style="color: #666666">2.5*</span>x[<span style="color: #666666">1</span>]<span style="color: #666666">**2</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">df</span>(x):
|
||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>array([x[<span style="color: #666666">0</span>], <span style="color: #666666">5*</span>x[<span style="color: #666666">1</span>]])
|
||
|
||
fig <span style="color: #666666">=</span> pt<span style="color: #666666">.</span>figure()
|
||
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>gca(projection<span style="color: #666666">=</span><span style="color: #BA2121">"3d"</span>)
|
||
|
||
xmesh, ymesh <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mgrid[<span style="color: #666666">-2</span>:<span style="color: #666666">2</span>:<span style="color: #666666">50j</span>,<span style="color: #666666">-2</span>:<span style="color: #666666">2</span>:<span style="color: #666666">50j</span>]
|
||
fmesh <span style="color: #666666">=</span> f(np<span style="color: #666666">.</span>array([xmesh, ymesh]))
|
||
ax<span style="color: #666666">.</span>plot_surface(xmesh, ymesh, fmesh)
|
||
</pre></div>
|
||
<p>
|
||
And then as countor plot
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>pt<span style="color: #666666">.</span>axis(<span style="color: #BA2121">"equal"</span>)
|
||
pt<span style="color: #666666">.</span>contour(xmesh, ymesh, fmesh)
|
||
guesses <span style="color: #666666">=</span> [np<span style="color: #666666">.</span>array([<span style="color: #666666">2</span>, <span style="color: #666666">2./5</span>])]
|
||
</pre></div>
|
||
<p>
|
||
Find guesses
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>x <span style="color: #666666">=</span> guesses[<span style="color: #666666">-1</span>]
|
||
s <span style="color: #666666">=</span> <span style="color: #666666">-</span>df(x)
|
||
</pre></div>
|
||
<p>
|
||
Run it!
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f1d</span>(alpha):
|
||
<span style="color: #008000; font-weight: bold">return</span> f(x <span style="color: #666666">+</span> alpha<span style="color: #666666">*</span>s)
|
||
|
||
alpha_opt <span style="color: #666666">=</span> sopt<span style="color: #666666">.</span>golden(f1d)
|
||
next_guess <span style="color: #666666">=</span> x <span style="color: #666666">+</span> alpha_opt <span style="color: #666666">*</span> s
|
||
guesses<span style="color: #666666">.</span>append(next_guess)
|
||
<span style="color: #008000; font-weight: bold">print</span>(next_guess)
|
||
</pre></div>
|
||
<p>
|
||
What happened?
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>pt<span style="color: #666666">.</span>axis(<span style="color: #BA2121">"equal"</span>)
|
||
pt<span style="color: #666666">.</span>contour(xmesh, ymesh, fmesh, <span style="color: #666666">50</span>)
|
||
it_array <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array(guesses)
|
||
pt<span style="color: #666666">.</span>plot(it_array<span style="color: #666666">.</span>T[<span style="color: #666666">0</span>], it_array<span style="color: #666666">.</span>T[<span style="color: #666666">1</span>], <span style="color: #BA2121">"x-"</span>)
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec25">Conjugate gradient </h2>
|
||
|
||
<p>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec26">Revisiting our first homework </h2>
|
||
|
||
<p>
|
||
We will use linear regression as a case study for the gradient descent
|
||
methods. Linear regression is a great test case for the gradient
|
||
descent methods discussed in the lectures since it has several
|
||
desirable properties such as:
|
||
|
||
<ol>
|
||
<li> An analytical solution (recall homework set 1).</li>
|
||
<li> The gradient can be computed analytically.</li>
|
||
<li> The cost function is convex which guarantees that gradient descent converges for small enough learning rates</li>
|
||
</ol>
|
||
|
||
We revisit the example from homework set 1 where we had
|
||
$$
|
||
y_i = 5x_i^2 + 0.1\xi_i, \ i=1,\cdots,100
|
||
$$
|
||
|
||
with \( x_i \in [0,1] \) chosen randomly with a uniform distribution. Additionally \( \xi_i \) represents stochastic noise chosen according to a normal distribution \( \cal {N}(0,1) \).
|
||
The linear regression model is given by
|
||
$$
|
||
h_\beta(x) = \hat{y} = \beta_0 + \beta_1 x,
|
||
$$
|
||
|
||
such that
|
||
$$
|
||
\hat{y}_i = \beta_0 + \beta_1 x_i.
|
||
$$
|
||
|
||
<p>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec27">Gradient descent example </h2>
|
||
|
||
<p>
|
||
Let \( \mathbf{y} = (y_1,\cdots,y_n)^T \), \( \mathbf{\hat{y}} = (\hat{y}_1,\cdots,\hat{y}_n)^T \) and \( \beta = (\beta_0, \beta_1)^T \)
|
||
|
||
<p>
|
||
It is convenient to write \( \mathbf{\hat{y}} = X\beta \) where \( X \in \mathbb{R}^{100 \times 2} \) is the design matrix given by
|
||
$$
|
||
X \equiv \begin{bmatrix}
|
||
1 & x_1 \\
|
||
\vdots & \vdots \\
|
||
1 & x_{100} & \\
|
||
\end{bmatrix}.
|
||
$$
|
||
|
||
The loss function is given by
|
||
$$
|
||
C(\beta) = ||X\beta-\mathbf{y}||^2 = ||X\beta||^2 - 2 \mathbf{y}^T X\beta + ||\mathbf{y}||^2 = \sum_{i=1}^{100} (\beta_0 + \beta_1 x_i)^2 - 2 y_i (\beta_0 + \beta_1 x_i) + y_i^2
|
||
$$
|
||
|
||
and we want to find \( \beta \) such that \( C(\beta) \) is minimized.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec28">The derivative of the cost/loss function </h2>
|
||
|
||
<p>
|
||
Computing \( \partial C(\beta) / \partial \beta_0 \) and \( \partial C(\beta) / \partial \beta_1 \) we can show that the gradient can be written as
|
||
$$
|
||
\nabla_{\beta} C(\beta) = (\partial C(\beta) / \partial \beta_0, \partial C(\beta) / \partial \beta_1)^T = 2\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\
|
||
\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\
|
||
\end{bmatrix} = 2X^T(X\beta - \mathbf{y}),
|
||
$$
|
||
|
||
where \( X \) is the design matrix defined above.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec29">The Hessian matrix </h2>
|
||
The Hessian matrix of \( C(\beta) \) is given by
|
||
$$
|
||
\hat{H} \equiv \begin{bmatrix}
|
||
\frac{\partial^2 C(\beta)}{\partial \beta_0^2} & \frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} \\
|
||
\frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} & \frac{\partial^2 C(\beta)}{\partial \beta_1^2} & \\
|
||
\end{bmatrix} = 2X^T X.
|
||
$$
|
||
|
||
This result implies that \( C(\beta) \) is a convex function since the matrix \( X^T X \) always is positive semi-definite.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec30">Simple program </h2>
|
||
|
||
<p>
|
||
We can now write a program that minimizes \( C(\beta) \) using the gradient descent method with a constant learning rate \( \gamma \) according to
|
||
$$
|
||
\beta_{k+1} = \beta_k - \gamma \nabla_\beta C(\beta_k), \ k=0,1,\cdots
|
||
$$
|
||
|
||
<p>
|
||
We can use the expression we computed for the gradient and let use a
|
||
\( \beta_0 \) be chosen randomly and let \( \gamma = 0.001 \). Stop iterating
|
||
when \( ||\nabla_\beta C(\beta_k) || \leq \epsilon = 10^{-8} \).
|
||
|
||
<p>
|
||
And finally we can compare our solution for \( \beta \) with the analytic result given by
|
||
\( \beta= (X^TX)^{-1} X^T \mathbf{y} \).
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
|
||
<span style="color: #BA2121; font-style: italic">"""</span>
|
||
<span style="color: #BA2121; font-style: italic">The following setup is just a suggestion, feel free to write it the way you like.</span>
|
||
<span style="color: #BA2121; font-style: italic">"""</span>
|
||
|
||
<span style="color: #408080; font-style: italic">#Setup problem described in the exercise</span>
|
||
N <span style="color: #666666">=</span> <span style="color: #666666">100</span> <span style="color: #408080; font-style: italic">#Nr of datapoints</span>
|
||
M <span style="color: #666666">=</span> <span style="color: #666666">2</span> <span style="color: #408080; font-style: italic">#Nr of features</span>
|
||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(N) <span style="color: #408080; font-style: italic">#Uniformly generated x-values in [0,1]</span>
|
||
y <span style="color: #666666">=</span> <span style="color: #666666">5*</span>x<span style="color: #666666">**2</span> <span style="color: #666666">+</span> <span style="color: #666666">0.1*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(N)
|
||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones(N),x] <span style="color: #408080; font-style: italic">#Construct design matrix</span>
|
||
|
||
<span style="color: #408080; font-style: italic">#Compute beta according to normal equations to compare with GD solution</span>
|
||
Xt_X_inv <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>inv(np<span style="color: #666666">.</span>dot(X<span style="color: #666666">.</span>T,X))
|
||
Xt_y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>dot(X<span style="color: #666666">.</span>transpose(),y)
|
||
beta_NE <span style="color: #666666">=</span> np<span style="color: #666666">.</span>dot(Xt_X_inv,Xt_y)
|
||
<span style="color: #008000; font-weight: bold">print</span>(beta_NE)
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec31">Gradient Descent Example </h2>
|
||
|
||
<p>
|
||
Another simple example is here
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #408080; font-style: italic"># Importing various packages</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">mpl_toolkits.mplot3d</span> <span style="color: #008000; font-weight: bold">import</span> Axes3D
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">matplotlib</span> <span style="color: #008000; font-weight: bold">import</span> cm
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">matplotlib.ticker</span> <span style="color: #008000; font-weight: bold">import</span> LinearLocator, FormatStrFormatter
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">sys</span>
|
||
|
||
x <span style="color: #666666">=</span> <span style="color: #666666">2*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(<span style="color: #666666">100</span>,<span style="color: #666666">1</span>)
|
||
y <span style="color: #666666">=</span> <span style="color: #666666">4+3*</span>x<span style="color: #666666">+</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">100</span>,<span style="color: #666666">1</span>)
|
||
|
||
xb <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((<span style="color: #666666">100</span>,<span style="color: #666666">1</span>)), x]
|
||
beta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>inv(xb<span style="color: #666666">.</span>T<span style="color: #666666">.</span>dot(xb))<span style="color: #666666">.</span>dot(xb<span style="color: #666666">.</span>T)<span style="color: #666666">.</span>dot(y)
|
||
<span style="color: #008000; font-weight: bold">print</span>(beta_linreg)
|
||
beta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)
|
||
|
||
eta <span style="color: #666666">=</span> <span style="color: #666666">0.1</span>
|
||
Niterations <span style="color: #666666">=</span> <span style="color: #666666">1000</span>
|
||
m <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">for</span> <span style="color: #008000">iter</span> <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Niterations):
|
||
gradients <span style="color: #666666">=</span> <span style="color: #666666">2.0/</span>m<span style="color: #666666">*</span>xb<span style="color: #666666">.</span>T<span style="color: #666666">.</span>dot(xb<span style="color: #666666">.</span>dot(beta)<span style="color: #666666">-</span>y)
|
||
beta <span style="color: #666666">-=</span> eta<span style="color: #666666">*</span>gradients
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(beta)
|
||
xnew <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([[<span style="color: #666666">0</span>],[<span style="color: #666666">2</span>]])
|
||
xbnew <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)), xnew]
|
||
ypredict <span style="color: #666666">=</span> xbnew<span style="color: #666666">.</span>dot(beta)
|
||
ypredict2 <span style="color: #666666">=</span> xbnew<span style="color: #666666">.</span>dot(beta_linreg)
|
||
plt<span style="color: #666666">.</span>plot(xnew, ypredict, <span style="color: #BA2121">"r-"</span>)
|
||
plt<span style="color: #666666">.</span>plot(xnew, ypredict2, <span style="color: #BA2121">"b-"</span>)
|
||
plt<span style="color: #666666">.</span>plot(x, y ,<span style="color: #BA2121">'ro'</span>)
|
||
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">0</span>,<span style="color: #666666">2.0</span>,<span style="color: #666666">0</span>, <span style="color: #666666">15.0</span>])
|
||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r'$x$'</span>)
|
||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r'$y$'</span>)
|
||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r'Gradient descent example'</span>)
|
||
plt<span style="color: #666666">.</span>show()
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec32">And a corresponding example using <b>scikit-learn</b> </h2>
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #408080; font-style: italic"># Importing various packages</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> SGDRegressor
|
||
|
||
x <span style="color: #666666">=</span> <span style="color: #666666">2*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(<span style="color: #666666">100</span>,<span style="color: #666666">1</span>)
|
||
y <span style="color: #666666">=</span> <span style="color: #666666">4+3*</span>x<span style="color: #666666">+</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">100</span>,<span style="color: #666666">1</span>)
|
||
|
||
xb <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((<span style="color: #666666">100</span>,<span style="color: #666666">1</span>)), x]
|
||
beta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>inv(xb<span style="color: #666666">.</span>T<span style="color: #666666">.</span>dot(xb))<span style="color: #666666">.</span>dot(xb<span style="color: #666666">.</span>T)<span style="color: #666666">.</span>dot(y)
|
||
<span style="color: #008000; font-weight: bold">print</span>(beta_linreg)
|
||
sgdreg <span style="color: #666666">=</span> SGDRegressor(n_iter <span style="color: #666666">=</span> <span style="color: #666666">50</span>, penalty<span style="color: #666666">=</span><span style="color: #008000">None</span>, eta0<span style="color: #666666">=0.1</span>)
|
||
sgdreg<span style="color: #666666">.</span>fit(x,y<span style="color: #666666">.</span>ravel())
|
||
<span style="color: #008000; font-weight: bold">print</span>(sgdreg<span style="color: #666666">.</span>intercept_, sgdreg<span style="color: #666666">.</span>coef_)
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec33">Gradient descent and Ridge </h2>
|
||
|
||
<p>
|
||
We have also discussed Ridge regression where the loss function contains a regularized given by the \( L_2 \) norm of \( \beta \),
|
||
$$
|
||
C_{\text{ridge}}(\beta) = ||X\beta -\mathbf{y}||^2 + \lambda ||\beta||^2, \ \lambda \geq 0.
|
||
$$
|
||
|
||
<p>
|
||
In order to minimize \( C_{\text{ridge}}(\beta) \) using GD we only have adjust the gradient as follows
|
||
$$
|
||
\nabla_\beta C_{\text{ridge}}(\beta) = 2\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\
|
||
\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\
|
||
\end{bmatrix} + 2\lambda\begin{bmatrix} \beta_0 \\ \beta_1\end{bmatrix} = 2 (X^T(X\beta - \mathbf{y})+\lambda \beta).
|
||
$$
|
||
|
||
<p>
|
||
We can now extend our program to minimize \( C_{\text{ridge}}(\beta) \) using gradient descent and compare with the analytical solution given by
|
||
$$
|
||
\beta_{\text{ridge}} = \left(X^T X + \lambda I_{2 \times 2} \right)^{-1} X^T \mathbf{y},
|
||
$$
|
||
|
||
for \( \lambda = {0,1,10,50,100} \) (\( \lambda = 0 \) corresponds to ordinary least squares).
|
||
We can then compute \( ||\beta_{\text{ridge}}|| \) for each \( \lambda \).
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
|
||
<span style="color: #BA2121; font-style: italic">"""</span>
|
||
<span style="color: #BA2121; font-style: italic">The following setup is just a suggestion, feel free to write it the way you like.</span>
|
||
<span style="color: #BA2121; font-style: italic">"""</span>
|
||
|
||
<span style="color: #408080; font-style: italic">#Setup problem described in the exercise</span>
|
||
N <span style="color: #666666">=</span> <span style="color: #666666">100</span> <span style="color: #408080; font-style: italic">#Nr of datapoints</span>
|
||
M <span style="color: #666666">=</span> <span style="color: #666666">2</span> <span style="color: #408080; font-style: italic">#Nr of features</span>
|
||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(N)
|
||
y <span style="color: #666666">=</span> <span style="color: #666666">5*</span>x<span style="color: #666666">**2</span> <span style="color: #666666">+</span> <span style="color: #666666">0.1*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(N)
|
||
|
||
|
||
<span style="color: #408080; font-style: italic">#Compute analytic beta for Ridge regression </span>
|
||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones(N),x]
|
||
XT_X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>dot(X<span style="color: #666666">.</span>T,X)
|
||
|
||
l <span style="color: #666666">=</span> <span style="color: #666666">0.1</span> <span style="color: #408080; font-style: italic">#Ridge parameter lambda</span>
|
||
Id <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(XT_X<span style="color: #666666">.</span>shape[<span style="color: #666666">0</span>])
|
||
|
||
Z <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>inv(XT_X<span style="color: #666666">+</span>l<span style="color: #666666">*</span>Id)
|
||
beta_ridge <span style="color: #666666">=</span> np<span style="color: #666666">.</span>dot(Z,np<span style="color: #666666">.</span>dot(X<span style="color: #666666">.</span>T,y))
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(beta_ridge)
|
||
<span style="color: #008000; font-weight: bold">print</span>(np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>norm(beta_ridge)) <span style="color: #408080; font-style: italic">#||beta||</span>
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec34">Automatic differentiation </h2>
|
||
Python has tools for so-called <b>automatic differentiation</b>.
|
||
Consider the following example
|
||
$$
|
||
f(x) = \sin\left(2\pi x + x^2\right)
|
||
$$
|
||
|
||
which has the following derivative
|
||
$$
|
||
f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right)
|
||
$$
|
||
|
||
Using <b>autograd</b> we have
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
|
||
<span style="color: #408080; font-style: italic"># To do elementwise differentiation:</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> elementwise_grad <span style="color: #008000; font-weight: bold">as</span> egrad
|
||
|
||
<span style="color: #408080; font-style: italic"># To plot:</span>
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f</span>(x):
|
||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sin(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi<span style="color: #666666">*</span>x <span style="color: #666666">+</span> x<span style="color: #666666">**2</span>)
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f_grad_analytic</span>(x):
|
||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>cos(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi<span style="color: #666666">*</span>x <span style="color: #666666">+</span> x<span style="color: #666666">**2</span>)<span style="color: #666666">*</span>(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi <span style="color: #666666">+</span> <span style="color: #666666">2*</span>x)
|
||
|
||
<span style="color: #408080; font-style: italic"># Do the comparison:</span>
|
||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">0</span>,<span style="color: #666666">1</span>,<span style="color: #666666">1000</span>)
|
||
|
||
f_grad <span style="color: #666666">=</span> egrad(f)
|
||
|
||
computed <span style="color: #666666">=</span> f_grad(x)
|
||
analytic <span style="color: #666666">=</span> f_grad_analytic(x)
|
||
|
||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">'Derivative computed from Autograd compared with the analytical derivative'</span>)
|
||
plt<span style="color: #666666">.</span>plot(x,computed,label<span style="color: #666666">=</span><span style="color: #BA2121">'autograd'</span>)
|
||
plt<span style="color: #666666">.</span>plot(x,analytic,label<span style="color: #666666">=</span><span style="color: #BA2121">'analytic'</span>)
|
||
|
||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'x'</span>)
|
||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'y'</span>)
|
||
plt<span style="color: #666666">.</span>legend()
|
||
|
||
plt<span style="color: #666666">.</span>show()
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The max absolute difference is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(np<span style="color: #666666">.</span>max(np<span style="color: #666666">.</span>abs(computed <span style="color: #666666">-</span> analytic))))
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec35">Using autograd </h2>
|
||
|
||
<p>
|
||
Here we
|
||
experiment with what kind of functions Autograd is capable
|
||
of finding the gradient of. The following Python functions are just
|
||
meant to illustrate what Autograd can do, but please feel free to
|
||
experiment with other, possibly more complicated, functions as well.
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f1</span>(x):
|
||
<span style="color: #008000; font-weight: bold">return</span> x<span style="color: #666666">**3</span> <span style="color: #666666">+</span> <span style="color: #666666">1</span>
|
||
|
||
f1_grad <span style="color: #666666">=</span> grad(f1)
|
||
|
||
<span style="color: #408080; font-style: italic"># Remember to send in float as argument to the computed gradient from Autograd!</span>
|
||
a <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||
|
||
<span style="color: #408080; font-style: italic"># See the evaluated gradient at a using autograd:</span>
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The gradient of f1 evaluated at a = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> using autograd is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(a,f1_grad(a)))
|
||
|
||
<span style="color: #408080; font-style: italic"># Compare with the analytical derivative, that is f1'(x) = 3*x**2 </span>
|
||
grad_analytical <span style="color: #666666">=</span> <span style="color: #666666">3*</span>a<span style="color: #666666">**2</span>
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The gradient of f1 evaluated at a = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> by finding the analytic expression is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(a,grad_analytical))
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec36">Autograd with more complicated functions </h2>
|
||
|
||
<p>
|
||
To differentiate with respect to two (or more) arguments of a Python
|
||
function, Autograd need to know at which variable the function if
|
||
being differentiated with respect to.
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f2</span>(x1,x2):
|
||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">3*</span>x1<span style="color: #666666">**3</span> <span style="color: #666666">+</span> x2<span style="color: #666666">*</span>(x1 <span style="color: #666666">-</span> <span style="color: #666666">5</span>) <span style="color: #666666">+</span> <span style="color: #666666">1</span>
|
||
|
||
<span style="color: #408080; font-style: italic"># By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1</span>
|
||
f2_grad_x1 <span style="color: #666666">=</span> grad(f2,<span style="color: #666666">0</span>)
|
||
|
||
<span style="color: #408080; font-style: italic"># ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad</span>
|
||
f2_grad_x2 <span style="color: #666666">=</span> grad(f2,<span style="color: #666666">1</span>)
|
||
|
||
x1 <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||
x2 <span style="color: #666666">=</span> <span style="color: #666666">3.0</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"Evaluating at x1 = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">, x2 = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x1,x2))
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"-"</span><span style="color: #666666">*30</span>)
|
||
|
||
<span style="color: #408080; font-style: italic"># Compare with the analytical derivatives:</span>
|
||
|
||
<span style="color: #408080; font-style: italic"># Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:</span>
|
||
f2_grad_x1_analytical <span style="color: #666666">=</span> <span style="color: #666666">9*</span>x1<span style="color: #666666">**2</span> <span style="color: #666666">+</span> x2
|
||
|
||
<span style="color: #408080; font-style: italic"># Derivative of f2 w.r.t x2 is: x1 - 5:</span>
|
||
f2_grad_x2_analytical <span style="color: #666666">=</span> x1 <span style="color: #666666">-</span> <span style="color: #666666">5</span>
|
||
|
||
<span style="color: #408080; font-style: italic"># See the evaluated derivations:</span>
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The derivative of f2 w.r.t x1: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>( f2_grad_x1(x1,x2) ))
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The analytical derivative of f2 w.r.t x1: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>( f2_grad_x1(x1,x2) ))
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>()
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The derivative of f2 w.r.t x2: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>( f2_grad_x2(x1,x2) ))
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The analytical derivative of f2 w.r.t x2: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>( f2_grad_x2(x1,x2) ))
|
||
</pre></div>
|
||
<p>
|
||
Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec37">More complicated functions using the elements of their arguments directly </h2>
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f3</span>(x): <span style="color: #408080; font-style: italic"># Assumes x is an array of length 5 or higher</span>
|
||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">2*</span>x[<span style="color: #666666">0</span>] <span style="color: #666666">+</span> <span style="color: #666666">3*</span>x[<span style="color: #666666">1</span>] <span style="color: #666666">+</span> <span style="color: #666666">5*</span>x[<span style="color: #666666">2</span>] <span style="color: #666666">+</span> <span style="color: #666666">7*</span>x[<span style="color: #666666">3</span>] <span style="color: #666666">+</span> <span style="color: #666666">11*</span>x[<span style="color: #666666">4</span>]<span style="color: #666666">**2</span>
|
||
|
||
f3_grad <span style="color: #666666">=</span> grad(f3)
|
||
|
||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">0</span>,<span style="color: #666666">4</span>,<span style="color: #666666">5</span>)
|
||
|
||
<span style="color: #408080; font-style: italic"># Print the computed gradient:</span>
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The computed gradient of f3 is: "</span>, f3_grad(x))
|
||
|
||
<span style="color: #408080; font-style: italic"># The analytical gradient is: (2, 3, 5, 7, 22*x[4])</span>
|
||
f3_grad_analytical <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">2</span>, <span style="color: #666666">3</span>, <span style="color: #666666">5</span>, <span style="color: #666666">7</span>, <span style="color: #666666">22*</span>x[<span style="color: #666666">4</span>]])
|
||
|
||
<span style="color: #408080; font-style: italic"># Print the analytical gradient:</span>
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The analytical gradient of f3 is: "</span>, f3_grad_analytical)
|
||
</pre></div>
|
||
<p>
|
||
Note that in this case, when sending an array as input argument, the
|
||
output from Autograd is another array. This is the true gradient of
|
||
the function, as opposed to the function in the previous example. By
|
||
using arrays to represent the variables, the output from Autograd
|
||
might be easier to work with, as the output is closer to what one
|
||
could expect form a gradient-evaluting function.
|
||
|
||
<p>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec38">Functions using mathematical functions from Numpy </h2>
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f4</span>(x):
|
||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sqrt(<span style="color: #666666">1+</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> np<span style="color: #666666">.</span>exp(x) <span style="color: #666666">+</span> np<span style="color: #666666">.</span>sin(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi<span style="color: #666666">*</span>x)
|
||
|
||
f4_grad <span style="color: #666666">=</span> grad(f4)
|
||
|
||
x <span style="color: #666666">=</span> <span style="color: #666666">2.7</span>
|
||
|
||
<span style="color: #408080; font-style: italic"># Print the computed derivative:</span>
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The computed derivative of f4 at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f4_grad(x)))
|
||
|
||
<span style="color: #408080; font-style: italic"># The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi</span>
|
||
f4_grad_analytical <span style="color: #666666">=</span> x<span style="color: #666666">/</span>np<span style="color: #666666">.</span>sqrt(<span style="color: #666666">1</span> <span style="color: #666666">+</span> x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> np<span style="color: #666666">.</span>exp(x) <span style="color: #666666">+</span> np<span style="color: #666666">.</span>cos(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi<span style="color: #666666">*</span>x)<span style="color: #666666">*2*</span>np<span style="color: #666666">.</span>pi
|
||
|
||
<span style="color: #408080; font-style: italic"># Print the analytical gradient:</span>
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The analytical gradient of f4 at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f4_grad_analytical))
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec39">More autograd </h2>
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f5</span>(x):
|
||
<span style="color: #008000; font-weight: bold">if</span> x <span style="color: #666666">>=</span> <span style="color: #666666">0</span>:
|
||
<span style="color: #008000; font-weight: bold">return</span> x<span style="color: #666666">**2</span>
|
||
<span style="color: #008000; font-weight: bold">else</span>:
|
||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">-3*</span>x <span style="color: #666666">+</span> <span style="color: #666666">1</span>
|
||
|
||
f5_grad <span style="color: #666666">=</span> grad(f5)
|
||
|
||
x <span style="color: #666666">=</span> <span style="color: #666666">2.7</span>
|
||
|
||
<span style="color: #408080; font-style: italic"># Print the computed derivative:</span>
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The computed derivative of f5 at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f5_grad(x)))
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec40">And with loops </h2>
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f6_for</span>(x):
|
||
val <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">10</span>):
|
||
val <span style="color: #666666">=</span> val <span style="color: #666666">+</span> x<span style="color: #666666">**</span>i
|
||
<span style="color: #008000; font-weight: bold">return</span> val
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f6_while</span>(x):
|
||
val <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
||
i <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
||
<span style="color: #008000; font-weight: bold">while</span> i <span style="color: #666666"><</span> <span style="color: #666666">10</span>:
|
||
val <span style="color: #666666">=</span> val <span style="color: #666666">+</span> x<span style="color: #666666">**</span>i
|
||
i <span style="color: #666666">=</span> i <span style="color: #666666">+</span> <span style="color: #666666">1</span>
|
||
<span style="color: #008000; font-weight: bold">return</span> val
|
||
|
||
f6_for_grad <span style="color: #666666">=</span> grad(f6_for)
|
||
f6_while_grad <span style="color: #666666">=</span> grad(f6_while)
|
||
|
||
x <span style="color: #666666">=</span> <span style="color: #666666">0.5</span>
|
||
|
||
<span style="color: #408080; font-style: italic"># Print the computed derivaties of f6_for and f6_while</span>
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The computed derivative of f6_for at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f6_for_grad(x)))
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The computed derivative of f6_while at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f6_while_grad(x)))
|
||
</pre></div>
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
<span style="color: #408080; font-style: italic"># Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9</span>
|
||
<span style="color: #408080; font-style: italic"># The analytical derivative is: sum(i*x**(i-1)) </span>
|
||
f6_grad_analytical <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">10</span>):
|
||
f6_grad_analytical <span style="color: #666666">+=</span> i<span style="color: #666666">*</span>x<span style="color: #666666">**</span>(i<span style="color: #666666">-1</span>)
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The analytical derivative of f6 at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f6_grad_analytical))
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec41">Using recursion </h2>
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f7</span>(n): <span style="color: #408080; font-style: italic"># Assume that n is an integer</span>
|
||
<span style="color: #008000; font-weight: bold">if</span> n <span style="color: #666666">==</span> <span style="color: #666666">1</span> <span style="color: #AA22FF; font-weight: bold">or</span> n <span style="color: #666666">==</span> <span style="color: #666666">0</span>:
|
||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">1</span>
|
||
<span style="color: #008000; font-weight: bold">else</span>:
|
||
<span style="color: #008000; font-weight: bold">return</span> n<span style="color: #666666">*</span>f7(n<span style="color: #666666">-1</span>)
|
||
|
||
f7_grad <span style="color: #666666">=</span> grad(f7)
|
||
|
||
n <span style="color: #666666">=</span> <span style="color: #666666">2.0</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The computed derivative of f7 at n = </span><span style="color: #BB6688; font-weight: bold">%d</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(n,f7_grad(n)))
|
||
|
||
<span style="color: #408080; font-style: italic"># The function f7 is an implementation of the factorial of n.</span>
|
||
<span style="color: #408080; font-style: italic"># By using the product rule, one can find that the derivative is:</span>
|
||
|
||
f7_grad_analytical <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #008000">int</span>(n)<span style="color: #666666">-1</span>):
|
||
tmp <span style="color: #666666">=</span> <span style="color: #666666">1</span>
|
||
<span style="color: #008000; font-weight: bold">for</span> k <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #008000">int</span>(n)<span style="color: #666666">-1</span>):
|
||
<span style="color: #008000; font-weight: bold">if</span> k <span style="color: #666666">!=</span> i:
|
||
tmp <span style="color: #666666">*=</span> (n <span style="color: #666666">-</span> k)
|
||
f7_grad_analytical <span style="color: #666666">+=</span> tmp
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The analytical derivative of f7 at n = </span><span style="color: #BB6688; font-weight: bold">%d</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(n,f7_grad_analytical))
|
||
</pre></div>
|
||
<p>
|
||
Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec42">Unsupported functions </h2>
|
||
Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd.
|
||
|
||
<p>
|
||
Assigning a value to the variable being differentiated with respect to
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f8</span>(x): <span style="color: #408080; font-style: italic"># Assume x is an array</span>
|
||
x[<span style="color: #666666">2</span>] <span style="color: #666666">=</span> <span style="color: #666666">3</span>
|
||
<span style="color: #008000; font-weight: bold">return</span> x<span style="color: #666666">*2</span>
|
||
|
||
f8_grad <span style="color: #666666">=</span> grad(f8)
|
||
|
||
x <span style="color: #666666">=</span> <span style="color: #666666">8.4</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The derivative of f8 is:"</span>,f8_grad(x))
|
||
</pre></div>
|
||
<p>
|
||
Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec43">The syntax a.dot(b) when finding the dot product </h2>
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f9</span>(a): <span style="color: #408080; font-style: italic"># Assume a is an array with 2 elements</span>
|
||
b <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">1.0</span>,<span style="color: #666666">2.0</span>])
|
||
<span style="color: #008000; font-weight: bold">return</span> a<span style="color: #666666">.</span>dot(b)
|
||
|
||
f9_grad <span style="color: #666666">=</span> grad(f9)
|
||
|
||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">1.0</span>,<span style="color: #666666">0.0</span>])
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The derivative of f9 is:"</span>,f9_grad(x))
|
||
</pre></div>
|
||
<p>
|
||
Here we are told that the 'dot' function does not belong to Autograd's
|
||
version of a Numpy array. To overcome this, an alternative syntax
|
||
which also computed the dot product can be used:
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f9_alternative</span>(x): <span style="color: #408080; font-style: italic"># Assume a is an array with 2 elements</span>
|
||
b <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">1.0</span>,<span style="color: #666666">2.0</span>])
|
||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>dot(x,b) <span style="color: #408080; font-style: italic"># The same as x_1*b_1 + x_2*b_2</span>
|
||
|
||
f9_alternative_grad <span style="color: #666666">=</span> grad(f9_alternative)
|
||
|
||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">3.0</span>,<span style="color: #666666">0.0</span>])
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"The gradient of f9 is:"</span>,f9_alternative_grad(x))
|
||
|
||
<span style="color: #408080; font-style: italic"># The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively</span>
|
||
<span style="color: #408080; font-style: italic"># w.r.t x is (b_1, b_2).</span>
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec44">Recommended to avoid </h2>
|
||
The documentation recommends to avoid inplace operations such as
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>a <span style="color: #666666">+=</span> b
|
||
a <span style="color: #666666">-=</span> b
|
||
a<span style="color: #666666">*=</span> b
|
||
a <span style="color: #666666">/=</span>b
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec45">Stochastic Gradient Descent </h2>
|
||
|
||
<p>
|
||
Stochastic gradient descent (SGD) and variants thereof address some of
|
||
the shortcomings of the Gradient descent method discussed above.
|
||
|
||
<p>
|
||
The underlying idea of SGD comes from the observation that the cost
|
||
function, which we want to minimize, can almost always be written as a
|
||
sum over \( n \) data points \( \{\mathbf{x}_i\}_{i=1}^n \),
|
||
$$
|
||
C(\mathbf{\beta}) = \sum_{i=1}^n c_i(\mathbf{x}_i,
|
||
\mathbf{\beta}).
|
||
$$
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec46">Computation of gradients </h2>
|
||
|
||
<p>
|
||
This in turn means that the gradient can be
|
||
computed as a sum over \( i \)-gradients
|
||
$$
|
||
\nabla_\beta C(\mathbf{\beta}) = \sum_i^n \nabla_\beta c_i(\mathbf{x}_i,
|
||
\mathbf{\beta}).
|
||
$$
|
||
|
||
<p>
|
||
Stochasticity/randomness is introduced by only taking the
|
||
gradient on a subset of the data called minibatches. If there are \( n \)
|
||
data points and the size of each minibatch is \( M \), there will be \( n/M \)
|
||
minibatches. We denote these minibatches by \( B_k \) where
|
||
\( k=1,\cdots,n/M \).
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec47">SGD example </h2>
|
||
As an example, suppose we have \( 10 \) data points \( (\mathbf{x}_1,\cdots, \mathbf{x}_{10}) \)
|
||
and we choose to have \( M=5 \) minibathces,
|
||
then each minibatch contains two data points. In particular we have
|
||
\( B_1 = (\mathbf{x}_1,\mathbf{x}_2), \cdots, B_5 =
|
||
(\mathbf{x}_9,\mathbf{x}_{10}) \). Note that if you choose \( M=1 \) you
|
||
have only a single batch with all data points and on the other extreme,
|
||
you may choose \( M=n \) resulting in a minibatch for each datapoint, i.e
|
||
\( B_k = \mathbf{x}_k \).
|
||
|
||
<p>
|
||
The idea is now to approximate the gradient by replacing the sum over
|
||
all data points with a sum over the data points in one the minibatches
|
||
picked at random in each gradient descent step
|
||
$$
|
||
\nabla_{\beta}
|
||
C(\mathbf{\beta}) = \sum_{i=1}^n \nabla_\beta c_i(\mathbf{x}_i,
|
||
\mathbf{\beta}) \rightarrow \sum_{i \in B_k}^n \nabla_\beta
|
||
c_i(\mathbf{x}_i, \mathbf{\beta}).
|
||
$$
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec48">The gradient step </h2>
|
||
|
||
<p>
|
||
Thus a gradient descent step now looks like
|
||
$$
|
||
\beta_{j+1} = \beta_j - \gamma_j \sum_{i \in B_k}^n \nabla_\beta c_i(\mathbf{x}_i,
|
||
\mathbf{\beta})
|
||
$$
|
||
|
||
<p>
|
||
where \( k \) is picked at random with equal
|
||
probability from \( [1,n/M] \). An iteration over the number of
|
||
minibathces (n/M) is commonly referred to as an epoch. Thus it is
|
||
typical to choose a number of epochs and for each epoch iterate over
|
||
the number of minibatches, as exemplified in the code below.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec49">Simple example code </h2>
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
|
||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span> <span style="color: #408080; font-style: italic">#100 datapoints </span>
|
||
M <span style="color: #666666">=</span> <span style="color: #666666">5</span> <span style="color: #408080; font-style: italic">#size of each minibatch</span>
|
||
m <span style="color: #666666">=</span> <span style="color: #008000">int</span>(n<span style="color: #666666">/</span>M) <span style="color: #408080; font-style: italic">#number of minibatches</span>
|
||
n_epochs <span style="color: #666666">=</span> <span style="color: #666666">10</span> <span style="color: #408080; font-style: italic">#number of epochs</span>
|
||
|
||
j <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
||
<span style="color: #008000; font-weight: bold">for</span> epoch <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,n_epochs<span style="color: #666666">+1</span>):
|
||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(m):
|
||
k <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randint(m) <span style="color: #408080; font-style: italic">#Pick the k-th minibatch at random</span>
|
||
<span style="color: #408080; font-style: italic">#Compute the gradient using the data in minibatch Bk</span>
|
||
<span style="color: #408080; font-style: italic">#Compute new suggestion for </span>
|
||
j <span style="color: #666666">+=</span> <span style="color: #666666">1</span>
|
||
</pre></div>
|
||
<p>
|
||
Taking the gradient only on a subset of the data has two important
|
||
benefits. First, it introduces randomness which decreases the chance
|
||
that our opmization scheme gets stuck in a local minima. Second, if
|
||
the size of the minibatches are small relative to the number of
|
||
datapoints (\( M < n \)), the computation of the gradient is much
|
||
cheaper since we sum over the datapoints in the \( k-th \) minibatch and not
|
||
all \( n \) datapoints.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec50">When do we stop? </h2>
|
||
|
||
<p>
|
||
A natural question is when do we stop the search for a new minimum?
|
||
One possibility is to compute the full gradient after a given number
|
||
of epochs and check if the norm of the gradient is smaller than some
|
||
threshold and stop if true. However, the condition that the gradient
|
||
is zero is valid also for local minima, so this would only tell us
|
||
that we are close to a local/global minimum. However, we could also
|
||
evaluate the cost function at this point, store the result and
|
||
continue the search. If the test kicks in at a later stage we can
|
||
compare the values of the cost function and keep the \( \beta \) that
|
||
gave the lowest value.
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec51">Slightly different approach </h2>
|
||
|
||
<p>
|
||
Another approach is to let the step length \( \gamma_j \) depend on the
|
||
number of epochs in such a way that it becomes very small after a
|
||
reasonable time such that we do not move at all.
|
||
|
||
<p>
|
||
As an example, let \( e = 0,1,2,3,\cdots \) denote the current epoch and let \( t_0, t_1 > 0 \) be two fixed numbers. Furthermore, let \( t = e \cdot m + i \) where \( m \) is the number of minibatches and \( i=0,\cdots,m-1 \). Then the function $$\gamma_j(t; t_0, t_1) = \frac{t_0}{t+t_1} $$ goes to zero as the number of epochs gets large. I.e. we start with a step length \( \gamma_j (0; t_0, t_1) = t_0/t_1 \) which decays in <em>time</em> \( t \).
|
||
|
||
<p>
|
||
In this way we can fix the number of epochs, compute \( \beta \) and
|
||
evaluate the cost function at the end. Repeating the computation will
|
||
give a different result since the scheme is random by design. Then we
|
||
pick the final \( \beta \) that gives the lowest value of the cost
|
||
function.
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">step_length</span>(t,t0,t1):
|
||
<span style="color: #008000; font-weight: bold">return</span> t0<span style="color: #666666">/</span>(t<span style="color: #666666">+</span>t1)
|
||
|
||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span> <span style="color: #408080; font-style: italic">#100 datapoints </span>
|
||
M <span style="color: #666666">=</span> <span style="color: #666666">5</span> <span style="color: #408080; font-style: italic">#size of each minibatch</span>
|
||
m <span style="color: #666666">=</span> <span style="color: #008000">int</span>(n<span style="color: #666666">/</span>M) <span style="color: #408080; font-style: italic">#number of minibatches</span>
|
||
n_epochs <span style="color: #666666">=</span> <span style="color: #666666">500</span> <span style="color: #408080; font-style: italic">#number of epochs</span>
|
||
t0 <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||
t1 <span style="color: #666666">=</span> <span style="color: #666666">10</span>
|
||
|
||
gamma_j <span style="color: #666666">=</span> t0<span style="color: #666666">/</span>t1
|
||
j <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
||
<span style="color: #008000; font-weight: bold">for</span> epoch <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,n_epochs<span style="color: #666666">+1</span>):
|
||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(m):
|
||
k <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randint(m) <span style="color: #408080; font-style: italic">#Pick the k-th minibatch at random</span>
|
||
<span style="color: #408080; font-style: italic">#Compute the gradient using the data in minibatch Bk</span>
|
||
<span style="color: #408080; font-style: italic">#Compute new suggestion for beta</span>
|
||
t <span style="color: #666666">=</span> epoch<span style="color: #666666">*</span>m<span style="color: #666666">+</span>i
|
||
gamma_j <span style="color: #666666">=</span> step_length(t,t0,t1)
|
||
j <span style="color: #666666">+=</span> <span style="color: #666666">1</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"gamma_j after </span><span style="color: #BB6688; font-weight: bold">%d</span><span style="color: #BA2121"> epochs: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span> <span style="color: #666666">%</span> (n_epochs,gamma_j))
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec52">Program for stochastic gradient </h2>
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #408080; font-style: italic"># Importing various packages</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">math</span> <span style="color: #008000; font-weight: bold">import</span> exp, sqrt
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> SGDRegressor
|
||
|
||
x <span style="color: #666666">=</span> <span style="color: #666666">2*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(<span style="color: #666666">100</span>,<span style="color: #666666">1</span>)
|
||
y <span style="color: #666666">=</span> <span style="color: #666666">4+3*</span>x<span style="color: #666666">+</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">100</span>,<span style="color: #666666">1</span>)
|
||
|
||
xb <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((<span style="color: #666666">100</span>,<span style="color: #666666">1</span>)), x]
|
||
theta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>inv(xb<span style="color: #666666">.</span>T<span style="color: #666666">.</span>dot(xb))<span style="color: #666666">.</span>dot(xb<span style="color: #666666">.</span>T)<span style="color: #666666">.</span>dot(y)
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"Own inversion"</span>)
|
||
<span style="color: #008000; font-weight: bold">print</span>(theta_linreg)
|
||
sgdreg <span style="color: #666666">=</span> SGDRegressor(n_iter <span style="color: #666666">=</span> <span style="color: #666666">50</span>, penalty<span style="color: #666666">=</span><span style="color: #008000">None</span>, eta0<span style="color: #666666">=0.1</span>)
|
||
sgdreg<span style="color: #666666">.</span>fit(x,y<span style="color: #666666">.</span>ravel())
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"sgdreg from scikit"</span>)
|
||
<span style="color: #008000; font-weight: bold">print</span>(sgdreg<span style="color: #666666">.</span>intercept_, sgdreg<span style="color: #666666">.</span>coef_)
|
||
|
||
|
||
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)
|
||
|
||
eta <span style="color: #666666">=</span> <span style="color: #666666">0.1</span>
|
||
Niterations <span style="color: #666666">=</span> <span style="color: #666666">1000</span>
|
||
m <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">for</span> <span style="color: #008000">iter</span> <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Niterations):
|
||
gradients <span style="color: #666666">=</span> <span style="color: #666666">2.0/</span>m<span style="color: #666666">*</span>xb<span style="color: #666666">.</span>T<span style="color: #666666">.</span>dot(xb<span style="color: #666666">.</span>dot(theta)<span style="color: #666666">-</span>y)
|
||
theta <span style="color: #666666">-=</span> eta<span style="color: #666666">*</span>gradients
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"theta frm own gd"</span>)
|
||
<span style="color: #008000; font-weight: bold">print</span>(theta)
|
||
|
||
xnew <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([[<span style="color: #666666">0</span>],[<span style="color: #666666">2</span>]])
|
||
xbnew <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)), xnew]
|
||
ypredict <span style="color: #666666">=</span> xbnew<span style="color: #666666">.</span>dot(theta)
|
||
ypredict2 <span style="color: #666666">=</span> xbnew<span style="color: #666666">.</span>dot(theta_linreg)
|
||
|
||
|
||
n_epochs <span style="color: #666666">=</span> <span style="color: #666666">50</span>
|
||
t0, t1 <span style="color: #666666">=</span> <span style="color: #666666">5</span>, <span style="color: #666666">50</span>
|
||
m <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">learning_schedule</span>(t):
|
||
<span style="color: #008000; font-weight: bold">return</span> t0<span style="color: #666666">/</span>(t<span style="color: #666666">+</span>t1)
|
||
|
||
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)
|
||
|
||
<span style="color: #008000; font-weight: bold">for</span> epoch <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n_epochs):
|
||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(m):
|
||
random_index <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randint(m)
|
||
xi <span style="color: #666666">=</span> xb[random_index:random_index<span style="color: #666666">+1</span>]
|
||
yi <span style="color: #666666">=</span> y[random_index:random_index<span style="color: #666666">+1</span>]
|
||
gradients <span style="color: #666666">=</span> <span style="color: #666666">2</span> <span style="color: #666666">*</span> xi<span style="color: #666666">.</span>T<span style="color: #666666">.</span>dot(xi<span style="color: #666666">.</span>dot(theta)<span style="color: #666666">-</span>yi)
|
||
eta <span style="color: #666666">=</span> learning_schedule(epoch<span style="color: #666666">*</span>m<span style="color: #666666">+</span>i)
|
||
theta <span style="color: #666666">=</span> theta <span style="color: #666666">-</span> eta<span style="color: #666666">*</span>gradients
|
||
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">"theta from own sdg"</span>)
|
||
<span style="color: #008000; font-weight: bold">print</span>(theta)
|
||
|
||
|
||
|
||
|
||
|
||
|
||
plt<span style="color: #666666">.</span>plot(xnew, ypredict, <span style="color: #BA2121">"r-"</span>)
|
||
plt<span style="color: #666666">.</span>plot(xnew, ypredict2, <span style="color: #BA2121">"b-"</span>)
|
||
plt<span style="color: #666666">.</span>plot(x, y ,<span style="color: #BA2121">'ro'</span>)
|
||
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">0</span>,<span style="color: #666666">2.0</span>,<span style="color: #666666">0</span>, <span style="color: #666666">15.0</span>])
|
||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r'$x$'</span>)
|
||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r'$y$'</span>)
|
||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r'Random numbers '</span>)
|
||
plt<span style="color: #666666">.</span>show()
|
||
</pre></div>
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec53">Momentum based methods </h2>
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec54">Conjugate gradient method </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
In the CG method we define so-called conjugate directions and two vectors
|
||
\( \hat{s} \) and \( \hat{t} \)
|
||
are said to be
|
||
conjugate if
|
||
$$
|
||
\begin{equation*}
|
||
\hat{s}^T\hat{A}\hat{t}= 0.
|
||
\end{equation*}
|
||
$$
|
||
|
||
The philosophy of the CG method is to perform searches in various conjugate directions
|
||
of our vectors \( \hat{x}_i \) obeying the above criterion, namely
|
||
$$
|
||
\begin{equation*}
|
||
\hat{x}_i^T\hat{A}\hat{x}_j= 0.
|
||
\end{equation*}
|
||
$$
|
||
|
||
Two vectors are conjugate if they are orthogonal with respect to
|
||
this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \).
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec55">Conjugate gradient method </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
An example is given by the eigenvectors of the matrix
|
||
$$
|
||
\begin{equation*}
|
||
\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j,
|
||
\end{equation*}
|
||
$$
|
||
|
||
which is zero unless \( i=j \).
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec56">Conjugate gradient method </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size
|
||
\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector
|
||
$$
|
||
\begin{equation*}
|
||
\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}.
|
||
\end{equation*}
|
||
$$
|
||
|
||
We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions.
|
||
Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution
|
||
$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely
|
||
|
||
$$
|
||
\begin{equation*}
|
||
\hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i.
|
||
\end{equation*}
|
||
$$
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec57">Conjugate gradient method </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
The coefficients are given by
|
||
$$
|
||
\begin{equation*}
|
||
\mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}.
|
||
\end{equation*}
|
||
$$
|
||
|
||
Multiplying with \( \hat{p}_k^T \) from the left gives
|
||
|
||
$$
|
||
\begin{equation*}
|
||
\hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b},
|
||
\end{equation*}
|
||
$$
|
||
|
||
and we can define the coefficients \( \alpha_k \) as
|
||
|
||
$$
|
||
\begin{equation*}
|
||
\alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k}
|
||
\end{equation*}
|
||
$$
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec58">Conjugate gradient method and iterations </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
|
||
<p>
|
||
If we choose the conjugate vectors \( \hat{p}_k \) carefully,
|
||
then we may not need all of them to obtain a good approximation to the solution
|
||
\( \hat{x} \).
|
||
We want to regard the conjugate gradient method as an iterative method.
|
||
This will us to solve systems where \( n \) is so large that the direct
|
||
method would take too much time.
|
||
|
||
<p>
|
||
We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \).
|
||
We can assume without loss of generality that
|
||
$$
|
||
\begin{equation*}
|
||
\hat{x}_0=0,
|
||
\end{equation*}
|
||
$$
|
||
|
||
or consider the system
|
||
$$
|
||
\begin{equation*}
|
||
\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0,
|
||
\end{equation*}
|
||
$$
|
||
|
||
instead.
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec59">Conjugate gradient method </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form
|
||
$$
|
||
\begin{equation*}
|
||
f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n.
|
||
\end{equation*}
|
||
$$
|
||
|
||
This suggests taking the first basis vector \( \hat{p}_1 \)
|
||
to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \),
|
||
which equals
|
||
$$
|
||
\begin{equation*}
|
||
\hat{A}\hat{x}_0-\hat{b},
|
||
\end{equation*}
|
||
$$
|
||
|
||
and
|
||
\( \hat{x}_0=0 \) it is equal \( -\hat{b} \).
|
||
The other vectors in the basis will be conjugate to the gradient,
|
||
hence the name conjugate gradient method.
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec60">Conjugate gradient method </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
Let \( \hat{r}_k \) be the residual at the \( k \)-th step:
|
||
$$
|
||
\begin{equation*}
|
||
\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k.
|
||
\end{equation*}
|
||
$$
|
||
|
||
Note that \( \hat{r}_k \) is the negative gradient of \( f \) at
|
||
\( \hat{x}=\hat{x}_k \),
|
||
so the gradient descent method would be to move in the direction \( \hat{r}_k \).
|
||
Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other,
|
||
so we take the direction closest to the gradient \( \hat{r}_k \)
|
||
under the conjugacy constraint.
|
||
This gives the following expression
|
||
$$
|
||
\begin{equation*}
|
||
\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k.
|
||
\end{equation*}
|
||
$$
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec61">Conjugate gradient method </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
We can also compute the residual iteratively as
|
||
$$
|
||
\begin{equation*}
|
||
\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1},
|
||
\end{equation*}
|
||
$$
|
||
|
||
which equals
|
||
$$
|
||
\begin{equation*}
|
||
\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k),
|
||
\end{equation*}
|
||
$$
|
||
|
||
or
|
||
$$
|
||
\begin{equation*}
|
||
(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k,
|
||
\end{equation*}
|
||
$$
|
||
|
||
which gives
|
||
|
||
$$
|
||
\begin{equation*}
|
||
\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k},
|
||
\end{equation*}
|
||
$$
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec62">Simple implementation of the Conjugate gradient algorithm </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
<p>
|
||
|
||
<!-- code=c++ (!bc cppcod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span> Vector <span style="color: #0000FF">ConjugateGradient</span>(Matrix A, Vector b, Vector x0){
|
||
<span style="color: #B00040">int</span> dim <span style="color: #666666">=</span> x0.Dimension();
|
||
<span style="color: #008000; font-weight: bold">const</span> <span style="color: #B00040">double</span> tolerance <span style="color: #666666">=</span> <span style="color: #666666">1.0e-14</span>;
|
||
Vector x(dim),r(dim),v(dim),z(dim);
|
||
<span style="color: #B00040">double</span> c,t,d;
|
||
|
||
x <span style="color: #666666">=</span> x0;
|
||
r <span style="color: #666666">=</span> b <span style="color: #666666">-</span> A<span style="color: #666666">*</span>x;
|
||
v <span style="color: #666666">=</span> r;
|
||
c <span style="color: #666666">=</span> dot(r,r);
|
||
<span style="color: #B00040">int</span> i <span style="color: #666666">=</span> <span style="color: #666666">0</span>; IterMax <span style="color: #666666">=</span> dim;
|
||
<span style="color: #008000; font-weight: bold">while</span>(i <span style="color: #666666"><=</span> IterMax){
|
||
z <span style="color: #666666">=</span> A<span style="color: #666666">*</span>v;
|
||
t <span style="color: #666666">=</span> c<span style="color: #666666">/</span>dot(v,z);
|
||
x <span style="color: #666666">=</span> x <span style="color: #666666">+</span> t<span style="color: #666666">*</span>v;
|
||
r <span style="color: #666666">=</span> r <span style="color: #666666">-</span> t<span style="color: #666666">*</span>z;
|
||
d <span style="color: #666666">=</span> dot(r,r);
|
||
<span style="color: #008000; font-weight: bold">if</span>(sqrt(d) <span style="color: #666666"><</span> tolerance)
|
||
<span style="color: #008000; font-weight: bold">break</span>;
|
||
v <span style="color: #666666">=</span> r <span style="color: #666666">+</span> (d<span style="color: #666666">/</span>c)<span style="color: #666666">*</span>v;
|
||
c <span style="color: #666666">=</span> d; i<span style="color: #666666">++</span>;
|
||
}
|
||
<span style="color: #008000; font-weight: bold">return</span> x;
|
||
}
|
||
</pre></div>
|
||
|
||
</div>
|
||
|
||
|
||
<p>
|
||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||
|
||
<h2 id="___sec63">Broyden–Fletcher–Goldfarb–Shanno algorithm </h2>
|
||
<div class="alert alert-block alert-block alert-text-normal">
|
||
<b></b>
|
||
<p>
|
||
The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take.
|
||
|
||
<p>
|
||
The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage.
|
||
|
||
<p>
|
||
The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation
|
||
$$
|
||
B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}),
|
||
$$
|
||
|
||
<p>
|
||
where \( B_{k} \) is an approximation to the Hessian matrix, which is
|
||
updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \)
|
||
is the gradient of the function
|
||
evaluated at \( x_k \).
|
||
A line search in the direction \( p_k \) is then used to
|
||
find the next point \( x_{k+1} \) by minimising
|
||
$$
|
||
f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}),
|
||
$$
|
||
|
||
over the scalar \( \alpha > 0 \).
|
||
|
||
|
||
</div>
|
||
|
||
|
||
<p>
|
||
|
||
<!-- ------------------- end of main content --------------- -->
|
||
|
||
|
||
<center style="font-size:80%">
|
||
<!-- copyright --> © 1999-2018, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
|
||
</center>
|
||
|
||
|
||
</body>
|
||
</html>
|
||
|
||
|