cleaning up splines
This commit is contained in:
@@ -80,75 +80,73 @@ Automatically generated HTML file from DocOnce source
|
||||
None,
|
||||
'___sec24'),
|
||||
('Steepest descent example', 2, None, '___sec25'),
|
||||
('Conjugate gradient', 2, None, '___sec26'),
|
||||
('Revisiting our first homework', 2, None, '___sec27'),
|
||||
('Gradient descent example', 2, None, '___sec28'),
|
||||
('The derivative of the cost/loss function', 2, None, '___sec29'),
|
||||
('The Hessian matrix', 2, None, '___sec30'),
|
||||
('Simple program', 2, None, '___sec31'),
|
||||
('Gradient Descent Example', 2, None, '___sec32'),
|
||||
('Conjugate gradient method', 2, None, '___sec26'),
|
||||
('Conjugate gradient method', 2, None, '___sec27'),
|
||||
('Conjugate gradient method', 2, None, '___sec28'),
|
||||
('Conjugate gradient method', 2, None, '___sec29'),
|
||||
('Conjugate gradient method and iterations', 2, None, '___sec30'),
|
||||
('Conjugate gradient method', 2, None, '___sec31'),
|
||||
('Conjugate gradient method', 2, None, '___sec32'),
|
||||
('Conjugate gradient method', 2, None, '___sec33'),
|
||||
('Simple implementation of the Conjugate gradient algorithm',
|
||||
2,
|
||||
None,
|
||||
'___sec34'),
|
||||
('Broyden–Fletcher–Goldfarb–Shanno algorithm',
|
||||
2,
|
||||
None,
|
||||
'___sec35'),
|
||||
('Revisiting our first homework', 2, None, '___sec36'),
|
||||
('Gradient descent example', 2, None, '___sec37'),
|
||||
('The derivative of the cost/loss function', 2, None, '___sec38'),
|
||||
('The Hessian matrix', 2, None, '___sec39'),
|
||||
('Simple program', 2, None, '___sec40'),
|
||||
('Gradient Descent Example', 2, None, '___sec41'),
|
||||
('And a corresponding example using _scikit-learn_',
|
||||
2,
|
||||
None,
|
||||
'___sec33'),
|
||||
('Gradient descent and Ridge', 2, None, '___sec34'),
|
||||
('Automatic differentiation', 2, None, '___sec35'),
|
||||
('Using autograd', 2, None, '___sec36'),
|
||||
('Autograd with more complicated functions', 2, None, '___sec37'),
|
||||
'___sec42'),
|
||||
('Gradient descent and Ridge', 2, None, '___sec43'),
|
||||
('Automatic differentiation', 2, None, '___sec44'),
|
||||
('Using autograd', 2, None, '___sec45'),
|
||||
('Autograd with more complicated functions', 2, None, '___sec46'),
|
||||
('More complicated functions using the elements of their '
|
||||
'arguments directly',
|
||||
2,
|
||||
None,
|
||||
'___sec38'),
|
||||
'___sec47'),
|
||||
('Functions using mathematical functions from Numpy',
|
||||
2,
|
||||
None,
|
||||
'___sec39'),
|
||||
('More autograd', 2, None, '___sec40'),
|
||||
('And with loops', 2, None, '___sec41'),
|
||||
('Using recursion', 2, None, '___sec42'),
|
||||
('Unsupported functions', 2, None, '___sec43'),
|
||||
'___sec48'),
|
||||
('More autograd', 2, None, '___sec49'),
|
||||
('And with loops', 2, None, '___sec50'),
|
||||
('Using recursion', 2, None, '___sec51'),
|
||||
('Unsupported functions', 2, None, '___sec52'),
|
||||
('The syntax a.dot(b) when finding the dot product',
|
||||
2,
|
||||
None,
|
||||
'___sec44'),
|
||||
('Recommended to avoid', 2, None, '___sec45'),
|
||||
('Stochastic Gradient Descent', 2, None, '___sec46'),
|
||||
('Computation of gradients', 2, None, '___sec47'),
|
||||
('SGD example', 2, None, '___sec48'),
|
||||
('The gradient step', 2, None, '___sec49'),
|
||||
('Simple example code', 2, None, '___sec50'),
|
||||
('When do we stop?', 2, None, '___sec51'),
|
||||
('Slightly different approach', 2, None, '___sec52'),
|
||||
('Program for stochastic gradient', 2, None, '___sec53'),
|
||||
('Momentum based methods', 2, None, '___sec54'),
|
||||
('Conjugate gradient method', 2, None, '___sec55'),
|
||||
('Conjugate gradient method', 2, None, '___sec56'),
|
||||
('Conjugate gradient method', 2, None, '___sec57'),
|
||||
('Conjugate gradient method', 2, None, '___sec58'),
|
||||
('Conjugate gradient method and iterations', 2, None, '___sec59'),
|
||||
('Conjugate gradient method', 2, None, '___sec60'),
|
||||
('Conjugate gradient method', 2, None, '___sec61'),
|
||||
('Conjugate gradient method', 2, None, '___sec62'),
|
||||
('Simple implementation of the Conjugate gradient algorithm',
|
||||
2,
|
||||
None,
|
||||
'___sec63'),
|
||||
('Broyden–Fletcher–Goldfarb–Shanno algorithm',
|
||||
2,
|
||||
None,
|
||||
'___sec64'),
|
||||
'___sec53'),
|
||||
('Recommended to avoid', 2, None, '___sec54'),
|
||||
('Stochastic Gradient Descent', 2, None, '___sec55'),
|
||||
('Computation of gradients', 2, None, '___sec56'),
|
||||
('SGD example', 2, None, '___sec57'),
|
||||
('The gradient step', 2, None, '___sec58'),
|
||||
('Simple example code', 2, None, '___sec59'),
|
||||
('When do we stop?', 2, None, '___sec60'),
|
||||
('Slightly different approach', 2, None, '___sec61'),
|
||||
('Program for stochastic gradient', 2, None, '___sec62'),
|
||||
('Using gradient descent methods, limitations',
|
||||
2,
|
||||
None,
|
||||
'___sec65'),
|
||||
('Momentum based GD', 2, None, '___sec66'),
|
||||
('More on momentum based approaches', 2, None, '___sec67'),
|
||||
('Momentum parameter', 2, None, '___sec68'),
|
||||
('Second moment of the gradient', 2, None, '___sec69'),
|
||||
('RMS prop', 2, None, '___sec70'),
|
||||
('ADAM optimizer', 2, None, '___sec71'),
|
||||
('Practical tips', 2, None, '___sec72')]}
|
||||
'___sec63'),
|
||||
('Momentum based GD', 2, None, '___sec64'),
|
||||
('More on momentum based approaches', 2, None, '___sec65'),
|
||||
('Momentum parameter', 2, None, '___sec66'),
|
||||
('Second moment of the gradient', 2, None, '___sec67'),
|
||||
('RMS prop', 2, None, '___sec68'),
|
||||
('ADAM optimizer', 2, None, '___sec69'),
|
||||
('Practical tips', 2, None, '___sec70')]}
|
||||
end of tocinfo -->
|
||||
|
||||
<body>
|
||||
@@ -212,53 +210,51 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs024.html#___sec23" style="font-size: 80%;">Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs025.html#___sec24" style="font-size: 80%;">The routine for the steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs026.html#___sec25" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs027.html#___sec26" style="font-size: 80%;">Conjugate gradient</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs028.html#___sec27" style="font-size: 80%;">Revisiting our first homework</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs029.html#___sec28" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs030.html#___sec29" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs031.html#___sec30" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs032.html#___sec31" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs033.html#___sec32" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs034.html#___sec33" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs035.html#___sec34" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs036.html#___sec35" style="font-size: 80%;">Automatic differentiation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs037.html#___sec36" style="font-size: 80%;">Using autograd</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs038.html#___sec37" style="font-size: 80%;">Autograd with more complicated functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs039.html#___sec38" style="font-size: 80%;">More complicated functions using the elements of their arguments directly</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs040.html#___sec39" style="font-size: 80%;">Functions using mathematical functions from Numpy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs041.html#___sec40" style="font-size: 80%;">More autograd</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs042.html#___sec41" style="font-size: 80%;">And with loops</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs043.html#___sec42" style="font-size: 80%;">Using recursion</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs044.html#___sec43" style="font-size: 80%;">Unsupported functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs045.html#___sec44" style="font-size: 80%;">The syntax a.dot(b) when finding the dot product</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs046.html#___sec45" style="font-size: 80%;">Recommended to avoid</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs047.html#___sec46" style="font-size: 80%;">Stochastic Gradient Descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs048.html#___sec47" style="font-size: 80%;">Computation of gradients</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs049.html#___sec48" style="font-size: 80%;">SGD example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs050.html#___sec49" style="font-size: 80%;">The gradient step</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs051.html#___sec50" style="font-size: 80%;">Simple example code</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs052.html#___sec51" style="font-size: 80%;">When do we stop?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs053.html#___sec52" style="font-size: 80%;">Slightly different approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs054.html#___sec53" style="font-size: 80%;">Program for stochastic gradient</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs055.html#___sec54" style="font-size: 80%;">Momentum based methods</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs056.html#___sec55" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs057.html#___sec56" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs058.html#___sec57" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs059.html#___sec58" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs060.html#___sec59" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs061.html#___sec60" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs062.html#___sec61" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs063.html#___sec62" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#___sec63" style="font-size: 80%;">Simple implementation of the Conjugate gradient algorithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs065.html#___sec64" style="font-size: 80%;">Broyden–Fletcher–Goldfarb–Shanno algorithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs066.html#___sec65" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs067.html#___sec66" style="font-size: 80%;">Momentum based GD</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs068.html#___sec67" style="font-size: 80%;">More on momentum based approaches</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs069.html#___sec68" style="font-size: 80%;">Momentum parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs070.html#___sec69" style="font-size: 80%;">Second moment of the gradient</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs071.html#___sec70" style="font-size: 80%;">RMS prop</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs072.html#___sec71" style="font-size: 80%;">ADAM optimizer</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs073.html#___sec72" style="font-size: 80%;">Practical tips</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs027.html#___sec26" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs028.html#___sec27" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs029.html#___sec28" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs030.html#___sec29" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs031.html#___sec30" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs032.html#___sec31" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs033.html#___sec32" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs034.html#___sec33" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs035.html#___sec34" style="font-size: 80%;">Simple implementation of the Conjugate gradient algorithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs036.html#___sec35" style="font-size: 80%;">Broyden–Fletcher–Goldfarb–Shanno algorithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs037.html#___sec36" style="font-size: 80%;">Revisiting our first homework</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs038.html#___sec37" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs039.html#___sec38" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs040.html#___sec39" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs041.html#___sec40" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs042.html#___sec41" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs043.html#___sec42" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs044.html#___sec43" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs045.html#___sec44" style="font-size: 80%;">Automatic differentiation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs046.html#___sec45" style="font-size: 80%;">Using autograd</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs047.html#___sec46" style="font-size: 80%;">Autograd with more complicated functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs048.html#___sec47" style="font-size: 80%;">More complicated functions using the elements of their arguments directly</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs049.html#___sec48" style="font-size: 80%;">Functions using mathematical functions from Numpy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs050.html#___sec49" style="font-size: 80%;">More autograd</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs051.html#___sec50" style="font-size: 80%;">And with loops</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs052.html#___sec51" style="font-size: 80%;">Using recursion</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs053.html#___sec52" style="font-size: 80%;">Unsupported functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs054.html#___sec53" style="font-size: 80%;">The syntax a.dot(b) when finding the dot product</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs055.html#___sec54" style="font-size: 80%;">Recommended to avoid</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs056.html#___sec55" style="font-size: 80%;">Stochastic Gradient Descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs057.html#___sec56" style="font-size: 80%;">Computation of gradients</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs058.html#___sec57" style="font-size: 80%;">SGD example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs059.html#___sec58" style="font-size: 80%;">The gradient step</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs060.html#___sec59" style="font-size: 80%;">Simple example code</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs061.html#___sec60" style="font-size: 80%;">When do we stop?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs062.html#___sec61" style="font-size: 80%;">Slightly different approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs063.html#___sec62" style="font-size: 80%;">Program for stochastic gradient</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#___sec63" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs065.html#___sec64" style="font-size: 80%;">Momentum based GD</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs066.html#___sec65" style="font-size: 80%;">More on momentum based approaches</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs067.html#___sec66" style="font-size: 80%;">Momentum parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs068.html#___sec67" style="font-size: 80%;">Second moment of the gradient</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs069.html#___sec68" style="font-size: 80%;">RMS prop</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs070.html#___sec69" style="font-size: 80%;">ADAM optimizer</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._Splines-bs071.html#___sec70" style="font-size: 80%;">Practical tips</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -274,44 +270,17 @@ MathJax.Hub.Config({
|
||||
<a name="part0064"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="___sec63" class="anchor">Simple implementation of the Conjugate gradient algorithm </h2>
|
||||
<div class="panel panel-default">
|
||||
<div class="panel-body">
|
||||
<p> <!-- subsequent paragraphs come in larger fonts, so start with a paragraph -->
|
||||
<p>
|
||||
<h2 id="___sec63" class="anchor">Using gradient descent methods, limitations </h2>
|
||||
|
||||
<!-- code=c++ (!bc cppcod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span> Vector <span style="color: #0000FF">ConjugateGradient</span>(Matrix A, Vector b, Vector x0){
|
||||
<span style="color: #B00040">int</span> dim <span style="color: #666666">=</span> x0.Dimension();
|
||||
<span style="color: #008000; font-weight: bold">const</span> <span style="color: #B00040">double</span> tolerance <span style="color: #666666">=</span> <span style="color: #666666">1.0e-14</span>;
|
||||
Vector x(dim),r(dim),v(dim),z(dim);
|
||||
<span style="color: #B00040">double</span> c,t,d;
|
||||
<ul>
|
||||
<li> <b>Gradient descent (GD) finds local minima of our function</b>. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our energy function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.</li>
|
||||
<li> <b>GD is sensitive to initial conditions</b>. One consequence of the local nature of GD is that initial conditions matter. Depending on where one starts, one will end up at a different local minima. Therefore, it is very important to think about how one initializes the training process. This is true for GD as well as more complicated variants of GD.</li>
|
||||
<li> <b>Gradients are computationally expensive to calculate for large datasets</b>. In many cases in statistics and ML, the energy function is a sum of terms, with one term for each data point. For example, in linear regression, \( E \propto \sum_{i=1}^n (y_i - \mathbf{w}^T\cdot\mathbf{x}_i)^2 \); for logistic regression, the square error is replaced by the cross entropy. To calculate the gradient we have to sum over <em>all</em> \( n \) data points. Doing this at every GD step becomes extremely computationally expensive. An ingenious solution to this, is to calculate the gradients using small subsets of the data called "mini batches". This has the added benefit of introducing stochasticity into our algorithm.</li>
|
||||
<li> <b>GD is very sensitive to choices of learning rates</b>. GD is extremely sensitive to the choice of learning rates. If the learning rate is very small, the training process take an extremely long time. For larger learning rates, GD can diverge and give poor results. Furthermore, depending on what the local landscape looks like, we have to modify the learning rates to ensure convergence. Ideally, we would <em>adaptively</em> choose the learning rates to match the landscape.</li>
|
||||
<li> <b>GD treats all directions in parameter space uniformly.</b> Another major drawback of GD is that unlike Newton's method, the learning rate for GD is the same in all directions in parameter space. For this reason, the maximum learning rate is set by the behavior of the steepest direction and this can significantly slow down training. Ideally, we would like to take large steps in flat directions and small steps in steep directions. Since we are exploring rugged landscapes where curvatures change, this requires us to keep track of not only the gradient but second derivatives. The ideal scenario would be to calculate the Hessian but this proves to be too computationally expensive.</li>
|
||||
<li> GD can take exponential time to escape saddle points, even with random initialization. As we mentioned, GD is extremely sensitive to initial condition since it determines the particular local minimum GD would eventually reach. However, even with a good initialization scheme, through the introduction of randomness, GD can still take exponential time to escape saddle points.</li>
|
||||
</ul>
|
||||
|
||||
x <span style="color: #666666">=</span> x0;
|
||||
r <span style="color: #666666">=</span> b <span style="color: #666666">-</span> A<span style="color: #666666">*</span>x;
|
||||
v <span style="color: #666666">=</span> r;
|
||||
c <span style="color: #666666">=</span> dot(r,r);
|
||||
<span style="color: #B00040">int</span> i <span style="color: #666666">=</span> <span style="color: #666666">0</span>; IterMax <span style="color: #666666">=</span> dim;
|
||||
<span style="color: #008000; font-weight: bold">while</span>(i <span style="color: #666666"><=</span> IterMax){
|
||||
z <span style="color: #666666">=</span> A<span style="color: #666666">*</span>v;
|
||||
t <span style="color: #666666">=</span> c<span style="color: #666666">/</span>dot(v,z);
|
||||
x <span style="color: #666666">=</span> x <span style="color: #666666">+</span> t<span style="color: #666666">*</span>v;
|
||||
r <span style="color: #666666">=</span> r <span style="color: #666666">-</span> t<span style="color: #666666">*</span>z;
|
||||
d <span style="color: #666666">=</span> dot(r,r);
|
||||
<span style="color: #008000; font-weight: bold">if</span>(sqrt(d) <span style="color: #666666"><</span> tolerance)
|
||||
<span style="color: #008000; font-weight: bold">break</span>;
|
||||
v <span style="color: #666666">=</span> r <span style="color: #666666">+</span> (d<span style="color: #666666">/</span>c)<span style="color: #666666">*</span>v;
|
||||
c <span style="color: #666666">=</span> d; i<span style="color: #666666">++</span>;
|
||||
}
|
||||
<span style="color: #008000; font-weight: bold">return</span> x;
|
||||
}
|
||||
</pre></div>
|
||||
<p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
<ul class="pagination">
|
||||
@@ -334,8 +303,6 @@ MathJax.Hub.Config({
|
||||
<li><a href="._Splines-bs069.html">70</a></li>
|
||||
<li><a href="._Splines-bs070.html">71</a></li>
|
||||
<li><a href="._Splines-bs071.html">72</a></li>
|
||||
<li><a href="._Splines-bs072.html">73</a></li>
|
||||
<li><a href="._Splines-bs073.html">74</a></li>
|
||||
<li><a href="._Splines-bs065.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
Reference in New Issue
Block a user