869 lines
27 KiB
HTML
869 lines
27 KiB
HTML
<!--
|
|
Automatically generated HTML file from DocOnce source
|
|
(https://github.com/hplgit/doconce/)
|
|
-->
|
|
<html>
|
|
<head>
|
|
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
|
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
|
|
<meta name="description" content="Data Analysis and Machine Learning Lectures: Cubic Splines and Gradient Methods">
|
|
|
|
<title>Data Analysis and Machine Learning Lectures: Cubic Splines and Gradient Methods</title>
|
|
|
|
<!-- Bootstrap style: bootstrap -->
|
|
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
|
|
<!-- not necessary
|
|
<link href="https://netdna.bootstrapcdn.com/font-awesome/4.0.3/css/font-awesome.css" rel="stylesheet">
|
|
-->
|
|
|
|
<style type="text/css">
|
|
|
|
/* Add scrollbar to dropdown menus in bootstrap navigation bar */
|
|
.dropdown-menu {
|
|
height: auto;
|
|
max-height: 400px;
|
|
overflow-x: hidden;
|
|
}
|
|
|
|
/* Adds an invisible element before each target to offset for the navigation
|
|
bar */
|
|
.anchor::before {
|
|
content:"";
|
|
display:block;
|
|
height:50px; /* fixed header height for style bootstrap */
|
|
margin:-50px 0 0; /* negative fixed header height */
|
|
}
|
|
</style>
|
|
|
|
|
|
</head>
|
|
|
|
<!-- tocinfo
|
|
{'highest level': 2,
|
|
'sections': [('Cubic Splines', 2, None, '___sec0'),
|
|
('Splines', 2, None, '___sec1'),
|
|
('Splines', 2, None, '___sec2'),
|
|
('Splines', 2, None, '___sec3'),
|
|
('Splines', 2, None, '___sec4'),
|
|
('Splines', 2, None, '___sec5'),
|
|
('Splines', 2, None, '___sec6'),
|
|
('Splines', 2, None, '___sec7'),
|
|
('Splines', 2, None, '___sec8'),
|
|
('Splines', 2, None, '___sec9'),
|
|
('Conjugate gradient (CG) method', 2, None, '___sec10'),
|
|
('Conjugate gradient method', 2, None, '___sec11'),
|
|
("Conjugate gradient method, Newton's method first",
|
|
2,
|
|
None,
|
|
'___sec12'),
|
|
('Simple example and demonstration', 2, None, '___sec13'),
|
|
('Simple example and demonstration', 2, None, '___sec14'),
|
|
('Conjugate gradient method', 2, None, '___sec15'),
|
|
('Conjugate gradient method', 2, None, '___sec16'),
|
|
('Conjugate gradient method', 2, None, '___sec17'),
|
|
('Conjugate gradient method', 2, None, '___sec18'),
|
|
('Conjugate gradient method and iterations', 2, None, '___sec19'),
|
|
('Conjugate gradient method', 2, None, '___sec20'),
|
|
('Conjugate gradient method', 2, None, '___sec21'),
|
|
('Conjugate gradient method', 2, None, '___sec22')]}
|
|
end of tocinfo -->
|
|
|
|
<body>
|
|
|
|
|
|
|
|
<script type="text/x-mathjax-config">
|
|
MathJax.Hub.Config({
|
|
TeX: {
|
|
equationNumbers: { autoNumber: "AMS" },
|
|
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
|
|
}
|
|
});
|
|
</script>
|
|
<script type="text/javascript" async
|
|
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
|
|
</script>
|
|
|
|
|
|
|
|
|
|
<!-- Bootstrap navigation bar -->
|
|
<div class="navbar navbar-default navbar-fixed-top">
|
|
<div class="navbar-header">
|
|
<button type="button" class="navbar-toggle" data-toggle="collapse" data-target=".navbar-responsive-collapse">
|
|
<span class="icon-bar"></span>
|
|
<span class="icon-bar"></span>
|
|
<span class="icon-bar"></span>
|
|
</button>
|
|
<a class="navbar-brand" href="Splines-bs.html">Data Analysis and Machine Learning Lectures: Cubic Splines and Gradient Methods</a>
|
|
</div>
|
|
|
|
<div class="navbar-collapse collapse navbar-responsive-collapse">
|
|
<ul class="nav navbar-nav navbar-right">
|
|
<li class="dropdown">
|
|
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
|
|
<ul class="dropdown-menu">
|
|
<!-- navigation toc: --> <li><a href="#___sec0" style="font-size: 80%;">Cubic Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec1" style="font-size: 80%;">Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec2" style="font-size: 80%;">Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec3" style="font-size: 80%;">Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec4" style="font-size: 80%;">Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec5" style="font-size: 80%;">Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec6" style="font-size: 80%;">Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec7" style="font-size: 80%;">Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec8" style="font-size: 80%;">Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec9" style="font-size: 80%;">Splines</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec10" style="font-size: 80%;">Conjugate gradient (CG) method</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec11" style="font-size: 80%;">Conjugate gradient method</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec12" style="font-size: 80%;">Conjugate gradient method, Newton's method first</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec13" style="font-size: 80%;">Simple example and demonstration</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec14" style="font-size: 80%;">Simple example and demonstration</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec15" style="font-size: 80%;">Conjugate gradient method</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec16" style="font-size: 80%;">Conjugate gradient method</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec17" style="font-size: 80%;">Conjugate gradient method</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec18" style="font-size: 80%;">Conjugate gradient method</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec19" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec20" style="font-size: 80%;">Conjugate gradient method</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec21" style="font-size: 80%;">Conjugate gradient method</a></li>
|
|
<!-- navigation toc: --> <li><a href="#___sec22" style="font-size: 80%;">Conjugate gradient method</a></li>
|
|
|
|
</ul>
|
|
</li>
|
|
</ul>
|
|
</div>
|
|
</div>
|
|
</div> <!-- end of navigation bar -->
|
|
|
|
<div class="container">
|
|
|
|
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
|
|
|
<!-- ------------------- main content ---------------------- -->
|
|
|
|
|
|
|
|
<div class="jumbotron">
|
|
<center><h1>Data Analysis and Machine Learning Lectures: Cubic Splines and Gradient Methods</h1></center> <!-- document title -->
|
|
|
|
<p>
|
|
<!-- author(s): Morten Hjorth-Jensen -->
|
|
|
|
<center>
|
|
<b>Morten Hjorth-Jensen</b> [1, 2]
|
|
</center>
|
|
|
|
<p>
|
|
<!-- institution(s) -->
|
|
|
|
<center>[1] <b>Department of Physics, University of Oslo</b></center>
|
|
<center>[2] <b>Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University</b></center>
|
|
<br>
|
|
<p>
|
|
<center><h4>Jan 27, 2018</h4></center> <!-- date -->
|
|
<br>
|
|
<p>
|
|
<!-- potential-jumbotron-button -->
|
|
</div> <!-- end jumbotron -->
|
|
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec0" class="anchor">Cubic Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
Cubic spline interpolation is among one of the most used
|
|
methods for interpolating between data points where the arguments
|
|
are organized as ascending series. In the library program we supply
|
|
such a function, based on the so-called cubic spline method to be
|
|
described below.
|
|
|
|
<p>
|
|
A spline function consists of polynomial pieces defined on
|
|
subintervals. The different subintervals are connected via
|
|
various continuity relations.
|
|
|
|
<p>
|
|
Assume we have at our disposal \( n+1 \) points \( x_0, x_1, \dots x_n \)
|
|
arranged so that \( x_0 < x_1 < x_2 < \dots x_{n-1} < x_n \) (such points are called
|
|
knots). A spline function \( s \) of degree \( k \) with \( n+1 \) knots is defined
|
|
as follows
|
|
|
|
<ul>
|
|
<li> On every subinterval \( [x_{i-1},x_i) \) <em>s</em> is a polynomial of degree \( \le k \).</li>
|
|
<li> \( s \) has \( k-1 \) continuous derivatives in the whole interval \( [x_0,x_n] \).</li>
|
|
</ul>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec1" class="anchor">Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
As an example, consider a spline function of degree \( k=1 \) defined as follows
|
|
$$
|
|
s(x)=\begin{bmatrix} s_0(x)=a_0x+b_0 & x\in [x_0, x_1) \\
|
|
s_1(x)=a_1x+b_1 & x\in [x_1, x_2) \\
|
|
\dots & \dots \\
|
|
s_{n-1}(x)=a_{n-1}x+b_{n-1} & x\in
|
|
[x_{n-1}, x_n] \end{bmatrix}.
|
|
$$
|
|
|
|
In this case the polynomial consists of series of straight lines
|
|
connected to each other at every endpoint. The number of continuous
|
|
derivatives is then \( k-1=0 \), as expected when we deal with straight lines.
|
|
Such a polynomial is quite easy to construct given
|
|
\( n+1 \) points \( x_0, x_1, \dots x_n \) and their corresponding
|
|
function values.
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec2" class="anchor">Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
The most commonly used spline function is the one with \( k=3 \), the so-called
|
|
cubic spline function.
|
|
Assume that we have in adddition to the \( n+1 \) knots a series of
|
|
functions values \( y_0=f(x_0), y_1=f(x_1), \dots y_n=f(x_n) \).
|
|
By definition, the polynomials \( s_{i-1} \) and \( s_i \)
|
|
are thence supposed to interpolate the same point \( i \), that is
|
|
$$
|
|
s_{i-1}(x_i)= y_i = s_i(x_i),
|
|
$$
|
|
|
|
with \( 1 \le i \le n-1 \). In total we have \( n \) polynomials of the
|
|
type
|
|
$$
|
|
s_i(x)=a_{i0}+a_{i1}x+a_{i2}x^2+a_{i2}x^3,
|
|
$$
|
|
|
|
yielding \( 4n \) coefficients to determine.
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec3" class="anchor">Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
Every subinterval provides in addition the \( 2n \) conditions
|
|
$$
|
|
y_i = s(x_i),
|
|
$$
|
|
|
|
and
|
|
$$
|
|
s(x_{i+1})= y_{i+1},
|
|
$$
|
|
|
|
to be fulfilled. If we also assume that \( s' \) and \( s'' \) are continuous,
|
|
then
|
|
$$
|
|
s'_{i-1}(x_i)= s'_i(x_i),
|
|
$$
|
|
|
|
yields \( n-1 \) conditions. Similarly,
|
|
$$
|
|
s''_{i-1}(x_i)= s''_i(x_i),
|
|
$$
|
|
|
|
results in additional \( n-1 \) conditions. In total we have \( 4n \) coefficients
|
|
and \( 4n-2 \) equations to determine them, leaving us with \( 2 \) degrees of
|
|
freedom to be determined.
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec4" class="anchor">Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
Using the last equation we define two values for the second derivative, namely
|
|
$$
|
|
s''_{i}(x_i)= f_i,
|
|
$$
|
|
|
|
and
|
|
$$
|
|
s''_{i}(x_{i+1})= f_{i+1},
|
|
$$
|
|
|
|
and setting up a straight line between \( f_i \) and \( f_{i+1} \) we have
|
|
$$
|
|
s_i''(x) = \frac{f_i}{x_{i+1}-x_i}(x_{i+1}-x)+
|
|
\frac{f_{i+1}}{x_{i+1}-x_i}(x-x_i),
|
|
$$
|
|
|
|
and integrating twice one obtains
|
|
$$
|
|
s_i(x) = \frac{f_i}{6(x_{i+1}-x_i)}(x_{i+1}-x)^3+
|
|
\frac{f_{i+1}}{6(x_{i+1}-x_i)}(x-x_i)^3
|
|
+c(x-x_i)+d(x_{i+1}-x).
|
|
$$
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec5" class="anchor">Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
Using the conditions \( s_i(x_i)=y_i \) and \( s_i(x_{i+1})=y_{i+1} \)
|
|
we can in turn determine the constants \( c \) and \( d \) resulting in
|
|
$$
|
|
\begin{align}
|
|
s_i(x) =&\frac{f_i}{6(x_{i+1}-x_i)}(x_{i+1}-x)^3+
|
|
\frac{f_{i+1}}{6(x_{i+1}-x_i)}(x-x_i)^3 \nonumber \\
|
|
+&(\frac{y_{i+1}}{x_{i+1}-x_i}-\frac{f_{i+1}(x_{i+1}-x_i)}{6})
|
|
(x-x_i)+
|
|
(\frac{y_{i}}{x_{i+1}-x_i}-\frac{f_{i}(x_{i+1}-x_i)}{6})
|
|
(x_{i+1}-x).
|
|
\label{_auto1}
|
|
\end{align}
|
|
$$
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec6" class="anchor">Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
How to determine the values of the second
|
|
derivatives \( f_{i} \) and \( f_{i+1} \)? We use the continuity assumption
|
|
of the first derivatives
|
|
$$
|
|
s'_{i-1}(x_i)= s'_i(x_i),
|
|
$$
|
|
|
|
and set \( x=x_i \). Defining \( h_i=x_{i+1}-x_i \) we obtain finally
|
|
the following expression
|
|
$$
|
|
h_{i-1}f_{i-1}+2(h_{i}+h_{i-1})f_i+h_if_{i+1}=
|
|
\frac{6}{h_i}(y_{i+1}-y_i)-\frac{6}{h_{i-1}}(y_{i}-y_{i-1}),
|
|
$$
|
|
|
|
and introducing the shorthands \( u_i=2(h_{i}+h_{i-1}) \),
|
|
\( v_i=\frac{6}{h_i}(y_{i+1}-y_i)-\frac{6}{h_{i-1}}(y_{i}-y_{i-1}) \),
|
|
we can reformulate the problem as a set of linear equations to be
|
|
solved through e.g., Gaussian elemination
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec7" class="anchor">Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
Gaussian elimination
|
|
$$
|
|
\begin{bmatrix} u_1 & h_1 &0 &\dots & & & & \\
|
|
h_1 & u_2 & h_2 &0 &\dots & & & \\
|
|
0 & h_2 & u_3 & h_3 &0 &\dots & & \\
|
|
\dots& & \dots &\dots &\dots &\dots &\dots & \\
|
|
&\dots & & &0 &h_{n-3} &u_{n-2} &h_{n-2} \\
|
|
& && & &0 &h_{n-2} &u_{n-1} \end{bmatrix}
|
|
\begin{bmatrix} f_1 \\
|
|
f_2 \\
|
|
f_3\\
|
|
\dots \\
|
|
f_{n-2} \\
|
|
f_{n-1} \end{bmatrix} =
|
|
\begin{bmatrix} v_1 \\
|
|
v_2 \\
|
|
v_3\\
|
|
\dots \\
|
|
v_{n-2}\\
|
|
v_{n-1} \end{bmatrix}.
|
|
$$
|
|
|
|
Note that this is a set of tridiagonal equations and can be solved
|
|
through only \( O(n) \) operations.
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec8" class="anchor">Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
The functions supplied in the program library are <em>spline</em> and <em>splint</em>.
|
|
In order to use cubic spline interpolation you need first to call
|
|
|
|
<p>
|
|
|
|
<!-- code=c++ (!bc cppcod) typeset with pygments style "default" -->
|
|
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>spline(<span style="color: #B00040">double</span> x[], <span style="color: #B00040">double</span> y[], <span style="color: #B00040">int</span> n, <span style="color: #B00040">double</span> yp1, <span style="color: #B00040">double</span> yp2, <span style="color: #B00040">double</span> y2[])
|
|
</pre></div>
|
|
<p>
|
|
This function takes as
|
|
input \( x[0,..,n - 1] \) and \( y[0,..,n - 1] \) containing a tabulation
|
|
\( y_i = f(x_i) \) with \( x_0 < x_1 < .. < x_{n - 1} \)
|
|
together with the
|
|
first derivatives of \( f(x) \) at \( x_0 \) and \( x_{n-1} \), respectively. Then the
|
|
function returns \( y2[0,..,n-1] \) which contains the second derivatives of
|
|
\( f(x_i) \) at each point \( x_i \). \( n \) is the number of points.
|
|
This function provides the cubic spline interpolation for all subintervals
|
|
and is called only once.
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec9" class="anchor">Splines </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
Thereafter, if you wish to make various interpolations, you need to call the function
|
|
<p>
|
|
|
|
<!-- code=c++ (!bc cppcod) typeset with pygments style "default" -->
|
|
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>splint(<span style="color: #B00040">double</span> x[], <span style="color: #B00040">double</span> y[], <span style="color: #B00040">double</span> y2a[], <span style="color: #B00040">int</span> n, <span style="color: #B00040">double</span> x, <span style="color: #B00040">double</span> <span style="color: #666666">*</span>y)
|
|
</pre></div>
|
|
<p>
|
|
which takes as input
|
|
the tabulated values \( x[0,..,n - 1] \) and \( y[0,..,n - 1] \) and the output
|
|
y2a[0,..,n - 1] from <em>spline</em>. It returns the value \( y \) corresponding
|
|
to the point \( x \).
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec10" class="anchor">Conjugate gradient (CG) method </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
The success of the CG method for finding solutions of non-linear problems is based
|
|
on the theory of conjugate gradients for linear systems of equations. It belongs
|
|
to the class of iterative methods for solving problems from linear algebra of the type
|
|
$$
|
|
\begin{equation*}
|
|
\hat{A}\hat{x} = \hat{b}.
|
|
\end{equation*}
|
|
$$
|
|
|
|
In the iterative process we end up with a problem like
|
|
|
|
$$
|
|
\begin{equation*}
|
|
\hat{r}= \hat{b}-\hat{A}\hat{x},
|
|
\end{equation*}
|
|
$$
|
|
|
|
where \( \hat{r} \) is the so-called residual or error in the iterative process.
|
|
|
|
<p>
|
|
When we have found the exact solution, \( \hat{r}=0 \).
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec11" class="anchor">Conjugate gradient method </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
|
|
<p>
|
|
The residual is zero when we reach the minimum of the quadratic equation
|
|
$$
|
|
\begin{equation*}
|
|
P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b},
|
|
\end{equation*}
|
|
$$
|
|
|
|
with the constraint that the matrix \( \hat{A} \) is positive definite and symmetric.
|
|
If we search for a minimum of the quantum mechanical variance, then the matrix
|
|
\( \hat{A} \), which is called the Hessian, is given by the second-derivative of the function we want to minimize. This quantity is always positive definite. In our case this corresponds normally to the second derivative of the energy.
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec12" class="anchor">Conjugate gradient method, Newton's method first </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
We seek the minimum of the energy or the variance as function of various variational parameters.
|
|
In our case we have thus a function \( f \) whose minimum we are seeking.
|
|
In Newton's method we set \( \nabla f = 0 \) and we can thus compute the next iteration point
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}-\hat{x}_i=\hat{A}^{-1}\nabla f(\hat{x}_i).
|
|
\end{equation*}
|
|
$$
|
|
|
|
Subtracting this equation from that of \( \hat{x}_{i+1} \) we have
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}_{i+1}-\hat{x}_i=\hat{A}^{-1}(\nabla f(\hat{x}_{i+1})-\nabla f(\hat{x}_i)).
|
|
\end{equation*}
|
|
$$
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec13" class="anchor">Simple example and demonstration </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
The function \( f \) can be either the energy or the variance. If we choose the energy then we have
|
|
$$
|
|
\begin{equation*}
|
|
\hat{\alpha}_{i+1}-\hat{\alpha}_i=\hat{A}^{-1}(\nabla E(\hat{\alpha}_{i+1})-\nabla E(\hat{\alpha}_i)).
|
|
\end{equation*}
|
|
$$
|
|
|
|
In the simple harmonic oscillator model, the gradient and the Hessian \( \hat{A} \) are
|
|
$$
|
|
\begin{equation*}
|
|
\frac{d\langle E_L[\alpha]\rangle}{d\alpha} = \alpha-\frac{1}{4\alpha^3}
|
|
\end{equation*}
|
|
$$
|
|
|
|
and a second derivative which is always positive (meaning that we find a minimum)
|
|
$$
|
|
\begin{equation*}
|
|
\hat{A}= \frac{d^2\langle E_L[\alpha]\rangle}{d\alpha^2} = 1+\frac{3}{4\alpha^4}
|
|
\end{equation*}
|
|
$$
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec14" class="anchor">Simple example and demonstration </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
We get then
|
|
$$
|
|
\begin{equation*}
|
|
\alpha_{i+1}=\frac{4}{3}\alpha_i-\frac{\alpha_i^4}{3\alpha_{i+1}^3},
|
|
\end{equation*}
|
|
$$
|
|
|
|
which can be rewritten as
|
|
$$
|
|
\begin{equation*}
|
|
\alpha_{i+1}^4-\frac{4}{3}\alpha_i\alpha_{i+1}^4+\frac{1}{3}\alpha_i^4.
|
|
\end{equation*}
|
|
$$
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec15" class="anchor">Conjugate gradient method </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
In the CG method we define so-called conjugate directions and two vectors
|
|
\( \hat{s} \) and \( \hat{t} \)
|
|
are said to be
|
|
conjugate if
|
|
$$
|
|
\begin{equation*}
|
|
\hat{s}^T\hat{A}\hat{t}= 0.
|
|
\end{equation*}
|
|
$$
|
|
|
|
The philosophy of the CG method is to perform searches in various conjugate directions
|
|
of our vectors \( \hat{x}_i \) obeying the above criterion, namely
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}_i^T\hat{A}\hat{x}_j= 0.
|
|
\end{equation*}
|
|
$$
|
|
|
|
Two vectors are conjugate if they are orthogonal with respect to
|
|
this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \).
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec16" class="anchor">Conjugate gradient method </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
An example is given by the eigenvectors of the matrix
|
|
$$
|
|
\begin{equation*}
|
|
\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j,
|
|
\end{equation*}
|
|
$$
|
|
|
|
which is zero unless \( i=j \).
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec17" class="anchor">Conjugate gradient method </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size
|
|
\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}.
|
|
\end{equation*}
|
|
$$
|
|
|
|
We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions.
|
|
Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution
|
|
$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely
|
|
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i.
|
|
\end{equation*}
|
|
$$
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec18" class="anchor">Conjugate gradient method </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
The coefficients are given by
|
|
$$
|
|
\begin{equation*}
|
|
\mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}.
|
|
\end{equation*}
|
|
$$
|
|
|
|
Multiplying with \( \hat{p}_k^T \) from the left gives
|
|
|
|
$$
|
|
\begin{equation*}
|
|
\hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b},
|
|
\end{equation*}
|
|
$$
|
|
|
|
and we can define the coefficients \( \alpha_k \) as
|
|
|
|
$$
|
|
\begin{equation*}
|
|
\alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k}
|
|
\end{equation*}
|
|
$$
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec19" class="anchor">Conjugate gradient method and iterations </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
|
|
<p>
|
|
If we choose the conjugate vectors \( \hat{p}_k \) carefully,
|
|
then we may not need all of them to obtain a good approximation to the solution
|
|
\( \hat{x} \).
|
|
We want to regard the conjugate gradient method as an iterative method.
|
|
This will us to solve systems where \( n \) is so large that the direct
|
|
method would take too much time.
|
|
|
|
<p>
|
|
We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \).
|
|
We can assume without loss of generality that
|
|
$$
|
|
\begin{equation*}
|
|
\hat{x}_0=0,
|
|
\end{equation*}
|
|
$$
|
|
|
|
or consider the system
|
|
$$
|
|
\begin{equation*}
|
|
\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0,
|
|
\end{equation*}
|
|
$$
|
|
|
|
instead.
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec20" class="anchor">Conjugate gradient method </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form
|
|
$$
|
|
\begin{equation*}
|
|
f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n.
|
|
\end{equation*}
|
|
$$
|
|
|
|
This suggests taking the first basis vector \( \hat{p}_1 \)
|
|
to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \),
|
|
which equals
|
|
$$
|
|
\begin{equation*}
|
|
\hat{A}\hat{x}_0-\hat{b},
|
|
\end{equation*}
|
|
$$
|
|
|
|
and
|
|
\( \hat{x}_0=0 \) it is equal \( -\hat{b} \).
|
|
The other vectors in the basis will be conjugate to the gradient,
|
|
hence the name conjugate gradient method.
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec21" class="anchor">Conjugate gradient method </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
Let \( \hat{r}_k \) be the residual at the \( k \)-th step:
|
|
$$
|
|
\begin{equation*}
|
|
\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k.
|
|
\end{equation*}
|
|
$$
|
|
|
|
Note that \( \hat{r}_k \) is the negative gradient of \( f \) at
|
|
\( \hat{x}=\hat{x}_k \),
|
|
so the gradient descent method would be to move in the direction \( \hat{r}_k \).
|
|
Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other,
|
|
so we take the direction closest to the gradient \( \hat{r}_k \)
|
|
under the conjugacy constraint.
|
|
This gives the following expression
|
|
$$
|
|
\begin{equation*}
|
|
\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k.
|
|
\end{equation*}
|
|
$$
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
<!-- !split -->
|
|
|
|
<h2 id="___sec22" class="anchor">Conjugate gradient method </h2>
|
|
<div class="panel panel-default">
|
|
<div class="panel-body">
|
|
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
|
We can also compute the residual iteratively as
|
|
$$
|
|
\begin{equation*}
|
|
\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1},
|
|
\end{equation*}
|
|
$$
|
|
|
|
which equals
|
|
$$
|
|
\begin{equation*}
|
|
\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k),
|
|
\end{equation*}
|
|
$$
|
|
|
|
or
|
|
$$
|
|
\begin{equation*}
|
|
(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k,
|
|
\end{equation*}
|
|
$$
|
|
|
|
which gives
|
|
|
|
$$
|
|
\begin{equation*}
|
|
\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k},
|
|
\end{equation*}
|
|
$$
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<p>
|
|
|
|
<!-- ------------------- end of main content --------------- -->
|
|
|
|
</div> <!-- end container -->
|
|
<!-- include javascript, jQuery *first* -->
|
|
<script src="https://ajax.googleapis.com/ajax/libs/jquery/1.10.2/jquery.min.js"></script>
|
|
<script src="https://netdna.bootstrapcdn.com/bootstrap/3.0.0/js/bootstrap.min.js"></script>
|
|
|
|
<!-- Bootstrap footer
|
|
<footer>
|
|
<a href="http://..."><img width="250" align=right src="http://..."></a>
|
|
</footer>
|
|
-->
|
|
|
|
|
|
<center style="font-size:80%">
|
|
<!-- copyright --> © 1999-2018, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
|
|
</center>
|
|
|
|
|
|
</body>
|
|
</html>
|
|
|
|
|