Files
FYS-STK4155/doc/pub/Splines/html/Splines-bs.html
T
2018-01-27 09:55:46 -05:00

869 lines
27 KiB
HTML

<!--
Automatically generated HTML file from DocOnce source
(https://github.com/hplgit/doconce/)
-->
<html>
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
<meta name="description" content="Data Analysis and Machine Learning Lectures: Cubic Splines and Gradient Methods">
<title>Data Analysis and Machine Learning Lectures: Cubic Splines and Gradient Methods</title>
<!-- Bootstrap style: bootstrap -->
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
<!-- not necessary
<link href="https://netdna.bootstrapcdn.com/font-awesome/4.0.3/css/font-awesome.css" rel="stylesheet">
-->
<style type="text/css">
/* Add scrollbar to dropdown menus in bootstrap navigation bar */
.dropdown-menu {
height: auto;
max-height: 400px;
overflow-x: hidden;
}
/* Adds an invisible element before each target to offset for the navigation
bar */
.anchor::before {
content:"";
display:block;
height:50px; /* fixed header height for style bootstrap */
margin:-50px 0 0; /* negative fixed header height */
}
</style>
</head>
<!-- tocinfo
{'highest level': 2,
'sections': [('Cubic Splines', 2, None, '___sec0'),
('Splines', 2, None, '___sec1'),
('Splines', 2, None, '___sec2'),
('Splines', 2, None, '___sec3'),
('Splines', 2, None, '___sec4'),
('Splines', 2, None, '___sec5'),
('Splines', 2, None, '___sec6'),
('Splines', 2, None, '___sec7'),
('Splines', 2, None, '___sec8'),
('Splines', 2, None, '___sec9'),
('Conjugate gradient (CG) method', 2, None, '___sec10'),
('Conjugate gradient method', 2, None, '___sec11'),
("Conjugate gradient method, Newton's method first",
2,
None,
'___sec12'),
('Simple example and demonstration', 2, None, '___sec13'),
('Simple example and demonstration', 2, None, '___sec14'),
('Conjugate gradient method', 2, None, '___sec15'),
('Conjugate gradient method', 2, None, '___sec16'),
('Conjugate gradient method', 2, None, '___sec17'),
('Conjugate gradient method', 2, None, '___sec18'),
('Conjugate gradient method and iterations', 2, None, '___sec19'),
('Conjugate gradient method', 2, None, '___sec20'),
('Conjugate gradient method', 2, None, '___sec21'),
('Conjugate gradient method', 2, None, '___sec22')]}
end of tocinfo -->
<body>
<script type="text/x-mathjax-config">
MathJax.Hub.Config({
TeX: {
equationNumbers: { autoNumber: "AMS" },
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
}
});
</script>
<script type="text/javascript" async
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
</script>
<!-- Bootstrap navigation bar -->
<div class="navbar navbar-default navbar-fixed-top">
<div class="navbar-header">
<button type="button" class="navbar-toggle" data-toggle="collapse" data-target=".navbar-responsive-collapse">
<span class="icon-bar"></span>
<span class="icon-bar"></span>
<span class="icon-bar"></span>
</button>
<a class="navbar-brand" href="Splines-bs.html">Data Analysis and Machine Learning Lectures: Cubic Splines and Gradient Methods</a>
</div>
<div class="navbar-collapse collapse navbar-responsive-collapse">
<ul class="nav navbar-nav navbar-right">
<li class="dropdown">
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
<ul class="dropdown-menu">
<!-- navigation toc: --> <li><a href="#___sec0" style="font-size: 80%;">Cubic Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec1" style="font-size: 80%;">Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec2" style="font-size: 80%;">Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec3" style="font-size: 80%;">Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec4" style="font-size: 80%;">Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec5" style="font-size: 80%;">Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec6" style="font-size: 80%;">Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec7" style="font-size: 80%;">Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec8" style="font-size: 80%;">Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec9" style="font-size: 80%;">Splines</a></li>
<!-- navigation toc: --> <li><a href="#___sec10" style="font-size: 80%;">Conjugate gradient (CG) method</a></li>
<!-- navigation toc: --> <li><a href="#___sec11" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="#___sec12" style="font-size: 80%;">Conjugate gradient method, Newton's method first</a></li>
<!-- navigation toc: --> <li><a href="#___sec13" style="font-size: 80%;">Simple example and demonstration</a></li>
<!-- navigation toc: --> <li><a href="#___sec14" style="font-size: 80%;">Simple example and demonstration</a></li>
<!-- navigation toc: --> <li><a href="#___sec15" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="#___sec16" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="#___sec17" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="#___sec18" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="#___sec19" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
<!-- navigation toc: --> <li><a href="#___sec20" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="#___sec21" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="#___sec22" style="font-size: 80%;">Conjugate gradient method</a></li>
</ul>
</li>
</ul>
</div>
</div>
</div> <!-- end of navigation bar -->
<div class="container">
<p>&nbsp;</p><p>&nbsp;</p><p>&nbsp;</p> <!-- add vertical space -->
<!-- ------------------- main content ---------------------- -->
<div class="jumbotron">
<center><h1>Data Analysis and Machine Learning Lectures: Cubic Splines and Gradient Methods</h1></center> <!-- document title -->
<p>
<!-- author(s): Morten Hjorth-Jensen -->
<center>
<b>Morten Hjorth-Jensen</b> [1, 2]
</center>
<p>
<!-- institution(s) -->
<center>[1] <b>Department of Physics, University of Oslo</b></center>
<center>[2] <b>Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University</b></center>
<br>
<p>
<center><h4>Jan 27, 2018</h4></center> <!-- date -->
<br>
<p>
<!-- potential-jumbotron-button -->
</div> <!-- end jumbotron -->
<!-- !split -->
<h2 id="___sec0" class="anchor">Cubic Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
Cubic spline interpolation is among one of the most used
methods for interpolating between data points where the arguments
are organized as ascending series. In the library program we supply
such a function, based on the so-called cubic spline method to be
described below.
<p>
A spline function consists of polynomial pieces defined on
subintervals. The different subintervals are connected via
various continuity relations.
<p>
Assume we have at our disposal \( n+1 \) points \( x_0, x_1, \dots x_n \)
arranged so that \( x_0 < x_1 < x_2 < \dots x_{n-1} < x_n \) (such points are called
knots). A spline function \( s \) of degree \( k \) with \( n+1 \) knots is defined
as follows
<ul>
<li> On every subinterval \( [x_{i-1},x_i) \) <em>s</em> is a polynomial of degree \( \le k \).</li>
<li> \( s \) has \( k-1 \) continuous derivatives in the whole interval \( [x_0,x_n] \).</li>
</ul>
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec1" class="anchor">Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
As an example, consider a spline function of degree \( k=1 \) defined as follows
$$
s(x)=\begin{bmatrix} s_0(x)=a_0x+b_0 & x\in [x_0, x_1) \\
s_1(x)=a_1x+b_1 & x\in [x_1, x_2) \\
\dots & \dots \\
s_{n-1}(x)=a_{n-1}x+b_{n-1} & x\in
[x_{n-1}, x_n] \end{bmatrix}.
$$
In this case the polynomial consists of series of straight lines
connected to each other at every endpoint. The number of continuous
derivatives is then \( k-1=0 \), as expected when we deal with straight lines.
Such a polynomial is quite easy to construct given
\( n+1 \) points \( x_0, x_1, \dots x_n \) and their corresponding
function values.
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec2" class="anchor">Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
The most commonly used spline function is the one with \( k=3 \), the so-called
cubic spline function.
Assume that we have in adddition to the \( n+1 \) knots a series of
functions values \( y_0=f(x_0), y_1=f(x_1), \dots y_n=f(x_n) \).
By definition, the polynomials \( s_{i-1} \) and \( s_i \)
are thence supposed to interpolate the same point \( i \), that is
$$
s_{i-1}(x_i)= y_i = s_i(x_i),
$$
with \( 1 \le i \le n-1 \). In total we have \( n \) polynomials of the
type
$$
s_i(x)=a_{i0}+a_{i1}x+a_{i2}x^2+a_{i2}x^3,
$$
yielding \( 4n \) coefficients to determine.
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec3" class="anchor">Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
Every subinterval provides in addition the \( 2n \) conditions
$$
y_i = s(x_i),
$$
and
$$
s(x_{i+1})= y_{i+1},
$$
to be fulfilled. If we also assume that \( s' \) and \( s'' \) are continuous,
then
$$
s'_{i-1}(x_i)= s'_i(x_i),
$$
yields \( n-1 \) conditions. Similarly,
$$
s''_{i-1}(x_i)= s''_i(x_i),
$$
results in additional \( n-1 \) conditions. In total we have \( 4n \) coefficients
and \( 4n-2 \) equations to determine them, leaving us with \( 2 \) degrees of
freedom to be determined.
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec4" class="anchor">Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
Using the last equation we define two values for the second derivative, namely
$$
s''_{i}(x_i)= f_i,
$$
and
$$
s''_{i}(x_{i+1})= f_{i+1},
$$
and setting up a straight line between \( f_i \) and \( f_{i+1} \) we have
$$
s_i''(x) = \frac{f_i}{x_{i+1}-x_i}(x_{i+1}-x)+
\frac{f_{i+1}}{x_{i+1}-x_i}(x-x_i),
$$
and integrating twice one obtains
$$
s_i(x) = \frac{f_i}{6(x_{i+1}-x_i)}(x_{i+1}-x)^3+
\frac{f_{i+1}}{6(x_{i+1}-x_i)}(x-x_i)^3
+c(x-x_i)+d(x_{i+1}-x).
$$
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec5" class="anchor">Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
Using the conditions \( s_i(x_i)=y_i \) and \( s_i(x_{i+1})=y_{i+1} \)
we can in turn determine the constants \( c \) and \( d \) resulting in
$$
\begin{align}
s_i(x) =&\frac{f_i}{6(x_{i+1}-x_i)}(x_{i+1}-x)^3+
\frac{f_{i+1}}{6(x_{i+1}-x_i)}(x-x_i)^3 \nonumber \\
+&(\frac{y_{i+1}}{x_{i+1}-x_i}-\frac{f_{i+1}(x_{i+1}-x_i)}{6})
(x-x_i)+
(\frac{y_{i}}{x_{i+1}-x_i}-\frac{f_{i}(x_{i+1}-x_i)}{6})
(x_{i+1}-x).
\label{_auto1}
\end{align}
$$
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec6" class="anchor">Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
How to determine the values of the second
derivatives \( f_{i} \) and \( f_{i+1} \)? We use the continuity assumption
of the first derivatives
$$
s'_{i-1}(x_i)= s'_i(x_i),
$$
and set \( x=x_i \). Defining \( h_i=x_{i+1}-x_i \) we obtain finally
the following expression
$$
h_{i-1}f_{i-1}+2(h_{i}+h_{i-1})f_i+h_if_{i+1}=
\frac{6}{h_i}(y_{i+1}-y_i)-\frac{6}{h_{i-1}}(y_{i}-y_{i-1}),
$$
and introducing the shorthands \( u_i=2(h_{i}+h_{i-1}) \),
\( v_i=\frac{6}{h_i}(y_{i+1}-y_i)-\frac{6}{h_{i-1}}(y_{i}-y_{i-1}) \),
we can reformulate the problem as a set of linear equations to be
solved through e.g., Gaussian elemination
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec7" class="anchor">Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
Gaussian elimination
$$
\begin{bmatrix} u_1 & h_1 &0 &\dots & & & & \\
h_1 & u_2 & h_2 &0 &\dots & & & \\
0 & h_2 & u_3 & h_3 &0 &\dots & & \\
\dots& & \dots &\dots &\dots &\dots &\dots & \\
&\dots & & &0 &h_{n-3} &u_{n-2} &h_{n-2} \\
& && & &0 &h_{n-2} &u_{n-1} \end{bmatrix}
\begin{bmatrix} f_1 \\
f_2 \\
f_3\\
\dots \\
f_{n-2} \\
f_{n-1} \end{bmatrix} =
\begin{bmatrix} v_1 \\
v_2 \\
v_3\\
\dots \\
v_{n-2}\\
v_{n-1} \end{bmatrix}.
$$
Note that this is a set of tridiagonal equations and can be solved
through only \( O(n) \) operations.
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec8" class="anchor">Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
The functions supplied in the program library are <em>spline</em> and <em>splint</em>.
In order to use cubic spline interpolation you need first to call
<p>
<!-- code=c++ (!bc cppcod) typeset with pygments style "default" -->
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>spline(<span style="color: #B00040">double</span> x[], <span style="color: #B00040">double</span> y[], <span style="color: #B00040">int</span> n, <span style="color: #B00040">double</span> yp1, <span style="color: #B00040">double</span> yp2, <span style="color: #B00040">double</span> y2[])
</pre></div>
<p>
This function takes as
input \( x[0,..,n - 1] \) and \( y[0,..,n - 1] \) containing a tabulation
\( y_i = f(x_i) \) with \( x_0 < x_1 < .. < x_{n - 1} \)
together with the
first derivatives of \( f(x) \) at \( x_0 \) and \( x_{n-1} \), respectively. Then the
function returns \( y2[0,..,n-1] \) which contains the second derivatives of
\( f(x_i) \) at each point \( x_i \). \( n \) is the number of points.
This function provides the cubic spline interpolation for all subintervals
and is called only once.
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec9" class="anchor">Splines </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
Thereafter, if you wish to make various interpolations, you need to call the function
<p>
<!-- code=c++ (!bc cppcod) typeset with pygments style "default" -->
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span>splint(<span style="color: #B00040">double</span> x[], <span style="color: #B00040">double</span> y[], <span style="color: #B00040">double</span> y2a[], <span style="color: #B00040">int</span> n, <span style="color: #B00040">double</span> x, <span style="color: #B00040">double</span> <span style="color: #666666">*</span>y)
</pre></div>
<p>
which takes as input
the tabulated values \( x[0,..,n - 1] \) and \( y[0,..,n - 1] \) and the output
y2a[0,..,n - 1] from <em>spline</em>. It returns the value \( y \) corresponding
to the point \( x \).
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec10" class="anchor">Conjugate gradient (CG) method </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
The success of the CG method for finding solutions of non-linear problems is based
on the theory of conjugate gradients for linear systems of equations. It belongs
to the class of iterative methods for solving problems from linear algebra of the type
$$
\begin{equation*}
\hat{A}\hat{x} = \hat{b}.
\end{equation*}
$$
In the iterative process we end up with a problem like
$$
\begin{equation*}
\hat{r}= \hat{b}-\hat{A}\hat{x},
\end{equation*}
$$
where \( \hat{r} \) is the so-called residual or error in the iterative process.
<p>
When we have found the exact solution, \( \hat{r}=0 \).
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec11" class="anchor">Conjugate gradient method </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
<p>
The residual is zero when we reach the minimum of the quadratic equation
$$
\begin{equation*}
P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b},
\end{equation*}
$$
with the constraint that the matrix \( \hat{A} \) is positive definite and symmetric.
If we search for a minimum of the quantum mechanical variance, then the matrix
\( \hat{A} \), which is called the Hessian, is given by the second-derivative of the function we want to minimize. This quantity is always positive definite. In our case this corresponds normally to the second derivative of the energy.
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec12" class="anchor">Conjugate gradient method, Newton's method first </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
We seek the minimum of the energy or the variance as function of various variational parameters.
In our case we have thus a function \( f \) whose minimum we are seeking.
In Newton's method we set \( \nabla f = 0 \) and we can thus compute the next iteration point
$$
\begin{equation*}
\hat{x}-\hat{x}_i=\hat{A}^{-1}\nabla f(\hat{x}_i).
\end{equation*}
$$
Subtracting this equation from that of \( \hat{x}_{i+1} \) we have
$$
\begin{equation*}
\hat{x}_{i+1}-\hat{x}_i=\hat{A}^{-1}(\nabla f(\hat{x}_{i+1})-\nabla f(\hat{x}_i)).
\end{equation*}
$$
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec13" class="anchor">Simple example and demonstration </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
The function \( f \) can be either the energy or the variance. If we choose the energy then we have
$$
\begin{equation*}
\hat{\alpha}_{i+1}-\hat{\alpha}_i=\hat{A}^{-1}(\nabla E(\hat{\alpha}_{i+1})-\nabla E(\hat{\alpha}_i)).
\end{equation*}
$$
In the simple harmonic oscillator model, the gradient and the Hessian \( \hat{A} \) are
$$
\begin{equation*}
\frac{d\langle E_L[\alpha]\rangle}{d\alpha} = \alpha-\frac{1}{4\alpha^3}
\end{equation*}
$$
and a second derivative which is always positive (meaning that we find a minimum)
$$
\begin{equation*}
\hat{A}= \frac{d^2\langle E_L[\alpha]\rangle}{d\alpha^2} = 1+\frac{3}{4\alpha^4}
\end{equation*}
$$
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec14" class="anchor">Simple example and demonstration </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
We get then
$$
\begin{equation*}
\alpha_{i+1}=\frac{4}{3}\alpha_i-\frac{\alpha_i^4}{3\alpha_{i+1}^3},
\end{equation*}
$$
which can be rewritten as
$$
\begin{equation*}
\alpha_{i+1}^4-\frac{4}{3}\alpha_i\alpha_{i+1}^4+\frac{1}{3}\alpha_i^4.
\end{equation*}
$$
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec15" class="anchor">Conjugate gradient method </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
In the CG method we define so-called conjugate directions and two vectors
\( \hat{s} \) and \( \hat{t} \)
are said to be
conjugate if
$$
\begin{equation*}
\hat{s}^T\hat{A}\hat{t}= 0.
\end{equation*}
$$
The philosophy of the CG method is to perform searches in various conjugate directions
of our vectors \( \hat{x}_i \) obeying the above criterion, namely
$$
\begin{equation*}
\hat{x}_i^T\hat{A}\hat{x}_j= 0.
\end{equation*}
$$
Two vectors are conjugate if they are orthogonal with respect to
this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \).
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec16" class="anchor">Conjugate gradient method </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
An example is given by the eigenvectors of the matrix
$$
\begin{equation*}
\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j,
\end{equation*}
$$
which is zero unless \( i=j \).
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec17" class="anchor">Conjugate gradient method </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size
\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector
$$
\begin{equation*}
\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}.
\end{equation*}
$$
We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions.
Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution
$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely
$$
\begin{equation*}
\hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i.
\end{equation*}
$$
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec18" class="anchor">Conjugate gradient method </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
The coefficients are given by
$$
\begin{equation*}
\mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}.
\end{equation*}
$$
Multiplying with \( \hat{p}_k^T \) from the left gives
$$
\begin{equation*}
\hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b},
\end{equation*}
$$
and we can define the coefficients \( \alpha_k \) as
$$
\begin{equation*}
\alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k}
\end{equation*}
$$
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec19" class="anchor">Conjugate gradient method and iterations </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
<p>
If we choose the conjugate vectors \( \hat{p}_k \) carefully,
then we may not need all of them to obtain a good approximation to the solution
\( \hat{x} \).
We want to regard the conjugate gradient method as an iterative method.
This will us to solve systems where \( n \) is so large that the direct
method would take too much time.
<p>
We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \).
We can assume without loss of generality that
$$
\begin{equation*}
\hat{x}_0=0,
\end{equation*}
$$
or consider the system
$$
\begin{equation*}
\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0,
\end{equation*}
$$
instead.
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec20" class="anchor">Conjugate gradient method </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form
$$
\begin{equation*}
f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n.
\end{equation*}
$$
This suggests taking the first basis vector \( \hat{p}_1 \)
to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \),
which equals
$$
\begin{equation*}
\hat{A}\hat{x}_0-\hat{b},
\end{equation*}
$$
and
\( \hat{x}_0=0 \) it is equal \( -\hat{b} \).
The other vectors in the basis will be conjugate to the gradient,
hence the name conjugate gradient method.
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec21" class="anchor">Conjugate gradient method </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
Let \( \hat{r}_k \) be the residual at the \( k \)-th step:
$$
\begin{equation*}
\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k.
\end{equation*}
$$
Note that \( \hat{r}_k \) is the negative gradient of \( f \) at
\( \hat{x}=\hat{x}_k \),
so the gradient descent method would be to move in the direction \( \hat{r}_k \).
Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other,
so we take the direction closest to the gradient \( \hat{r}_k \)
under the conjugacy constraint.
This gives the following expression
$$
\begin{equation*}
\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k.
\end{equation*}
$$
</div>
</div>
<p>
<!-- !split -->
<h2 id="___sec22" class="anchor">Conjugate gradient method </h2>
<div class="panel panel-default">
<div class="panel-body">
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
We can also compute the residual iteratively as
$$
\begin{equation*}
\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1},
\end{equation*}
$$
which equals
$$
\begin{equation*}
\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k),
\end{equation*}
$$
or
$$
\begin{equation*}
(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k,
\end{equation*}
$$
which gives
$$
\begin{equation*}
\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k},
\end{equation*}
$$
</div>
</div>
<p>
<!-- ------------------- end of main content --------------- -->
</div> <!-- end container -->
<!-- include javascript, jQuery *first* -->
<script src="https://ajax.googleapis.com/ajax/libs/jquery/1.10.2/jquery.min.js"></script>
<script src="https://netdna.bootstrapcdn.com/bootstrap/3.0.0/js/bootstrap.min.js"></script>
<!-- Bootstrap footer
<footer>
<a href="http://..."><img width="250" align=right src="http://..."></a>
</footer>
-->
<center style="font-size:80%">
<!-- copyright --> &copy; 1999-2018, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
</center>
</body>
</html>