cleaning up splines

This commit is contained in:
mhjensen
2018-10-12 05:02:36 +02:00
parent c71c008945
commit 652f00be7f
80 changed files with 9955 additions and 10285 deletions
+120 -124
View File
@@ -80,75 +80,73 @@ Automatically generated HTML file from DocOnce source
None,
'___sec24'),
('Steepest descent example', 2, None, '___sec25'),
('Conjugate gradient', 2, None, '___sec26'),
('Revisiting our first homework', 2, None, '___sec27'),
('Gradient descent example', 2, None, '___sec28'),
('The derivative of the cost/loss function', 2, None, '___sec29'),
('The Hessian matrix', 2, None, '___sec30'),
('Simple program', 2, None, '___sec31'),
('Gradient Descent Example', 2, None, '___sec32'),
('Conjugate gradient method', 2, None, '___sec26'),
('Conjugate gradient method', 2, None, '___sec27'),
('Conjugate gradient method', 2, None, '___sec28'),
('Conjugate gradient method', 2, None, '___sec29'),
('Conjugate gradient method and iterations', 2, None, '___sec30'),
('Conjugate gradient method', 2, None, '___sec31'),
('Conjugate gradient method', 2, None, '___sec32'),
('Conjugate gradient method', 2, None, '___sec33'),
('Simple implementation of the Conjugate gradient algorithm',
2,
None,
'___sec34'),
('BroydenFletcherGoldfarbShanno algorithm',
2,
None,
'___sec35'),
('Revisiting our first homework', 2, None, '___sec36'),
('Gradient descent example', 2, None, '___sec37'),
('The derivative of the cost/loss function', 2, None, '___sec38'),
('The Hessian matrix', 2, None, '___sec39'),
('Simple program', 2, None, '___sec40'),
('Gradient Descent Example', 2, None, '___sec41'),
('And a corresponding example using _scikit-learn_',
2,
None,
'___sec33'),
('Gradient descent and Ridge', 2, None, '___sec34'),
('Automatic differentiation', 2, None, '___sec35'),
('Using autograd', 2, None, '___sec36'),
('Autograd with more complicated functions', 2, None, '___sec37'),
'___sec42'),
('Gradient descent and Ridge', 2, None, '___sec43'),
('Automatic differentiation', 2, None, '___sec44'),
('Using autograd', 2, None, '___sec45'),
('Autograd with more complicated functions', 2, None, '___sec46'),
('More complicated functions using the elements of their '
'arguments directly',
2,
None,
'___sec38'),
'___sec47'),
('Functions using mathematical functions from Numpy',
2,
None,
'___sec39'),
('More autograd', 2, None, '___sec40'),
('And with loops', 2, None, '___sec41'),
('Using recursion', 2, None, '___sec42'),
('Unsupported functions', 2, None, '___sec43'),
'___sec48'),
('More autograd', 2, None, '___sec49'),
('And with loops', 2, None, '___sec50'),
('Using recursion', 2, None, '___sec51'),
('Unsupported functions', 2, None, '___sec52'),
('The syntax a.dot(b) when finding the dot product',
2,
None,
'___sec44'),
('Recommended to avoid', 2, None, '___sec45'),
('Stochastic Gradient Descent', 2, None, '___sec46'),
('Computation of gradients', 2, None, '___sec47'),
('SGD example', 2, None, '___sec48'),
('The gradient step', 2, None, '___sec49'),
('Simple example code', 2, None, '___sec50'),
('When do we stop?', 2, None, '___sec51'),
('Slightly different approach', 2, None, '___sec52'),
('Program for stochastic gradient', 2, None, '___sec53'),
('Momentum based methods', 2, None, '___sec54'),
('Conjugate gradient method', 2, None, '___sec55'),
('Conjugate gradient method', 2, None, '___sec56'),
('Conjugate gradient method', 2, None, '___sec57'),
('Conjugate gradient method', 2, None, '___sec58'),
('Conjugate gradient method and iterations', 2, None, '___sec59'),
('Conjugate gradient method', 2, None, '___sec60'),
('Conjugate gradient method', 2, None, '___sec61'),
('Conjugate gradient method', 2, None, '___sec62'),
('Simple implementation of the Conjugate gradient algorithm',
2,
None,
'___sec63'),
('BroydenFletcherGoldfarbShanno algorithm',
2,
None,
'___sec64'),
'___sec53'),
('Recommended to avoid', 2, None, '___sec54'),
('Stochastic Gradient Descent', 2, None, '___sec55'),
('Computation of gradients', 2, None, '___sec56'),
('SGD example', 2, None, '___sec57'),
('The gradient step', 2, None, '___sec58'),
('Simple example code', 2, None, '___sec59'),
('When do we stop?', 2, None, '___sec60'),
('Slightly different approach', 2, None, '___sec61'),
('Program for stochastic gradient', 2, None, '___sec62'),
('Using gradient descent methods, limitations',
2,
None,
'___sec65'),
('Momentum based GD', 2, None, '___sec66'),
('More on momentum based approaches', 2, None, '___sec67'),
('Momentum parameter', 2, None, '___sec68'),
('Second moment of the gradient', 2, None, '___sec69'),
('RMS prop', 2, None, '___sec70'),
('ADAM optimizer', 2, None, '___sec71'),
('Practical tips', 2, None, '___sec72')]}
'___sec63'),
('Momentum based GD', 2, None, '___sec64'),
('More on momentum based approaches', 2, None, '___sec65'),
('Momentum parameter', 2, None, '___sec66'),
('Second moment of the gradient', 2, None, '___sec67'),
('RMS prop', 2, None, '___sec68'),
('ADAM optimizer', 2, None, '___sec69'),
('Practical tips', 2, None, '___sec70')]}
end of tocinfo -->
<body>
@@ -212,53 +210,51 @@ MathJax.Hub.Config({
<!-- navigation toc: --> <li><a href="._Splines-bs024.html#___sec23" style="font-size: 80%;">Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs025.html#___sec24" style="font-size: 80%;">The routine for the steepest descent method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs026.html#___sec25" style="font-size: 80%;">Steepest descent example</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs027.html#___sec26" style="font-size: 80%;">Conjugate gradient</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs028.html#___sec27" style="font-size: 80%;">Revisiting our first homework</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs029.html#___sec28" style="font-size: 80%;">Gradient descent example</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs030.html#___sec29" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs031.html#___sec30" style="font-size: 80%;">The Hessian matrix</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs032.html#___sec31" style="font-size: 80%;">Simple program</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs033.html#___sec32" style="font-size: 80%;">Gradient Descent Example</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs034.html#___sec33" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs035.html#___sec34" style="font-size: 80%;">Gradient descent and Ridge</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs036.html#___sec35" style="font-size: 80%;">Automatic differentiation</a></li>
<!-- navigation toc: --> <li><a href="#___sec36" style="font-size: 80%;">Using autograd</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs038.html#___sec37" style="font-size: 80%;">Autograd with more complicated functions</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs039.html#___sec38" style="font-size: 80%;">More complicated functions using the elements of their arguments directly</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs040.html#___sec39" style="font-size: 80%;">Functions using mathematical functions from Numpy</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs041.html#___sec40" style="font-size: 80%;">More autograd</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs042.html#___sec41" style="font-size: 80%;">And with loops</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs043.html#___sec42" style="font-size: 80%;">Using recursion</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs044.html#___sec43" style="font-size: 80%;">Unsupported functions</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs045.html#___sec44" style="font-size: 80%;">The syntax a.dot(b) when finding the dot product</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs046.html#___sec45" style="font-size: 80%;">Recommended to avoid</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs047.html#___sec46" style="font-size: 80%;">Stochastic Gradient Descent</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs048.html#___sec47" style="font-size: 80%;">Computation of gradients</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs049.html#___sec48" style="font-size: 80%;">SGD example</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs050.html#___sec49" style="font-size: 80%;">The gradient step</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs051.html#___sec50" style="font-size: 80%;">Simple example code</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs052.html#___sec51" style="font-size: 80%;">When do we stop?</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs053.html#___sec52" style="font-size: 80%;">Slightly different approach</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs054.html#___sec53" style="font-size: 80%;">Program for stochastic gradient</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs055.html#___sec54" style="font-size: 80%;">Momentum based methods</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs056.html#___sec55" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs057.html#___sec56" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs058.html#___sec57" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs059.html#___sec58" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs060.html#___sec59" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs061.html#___sec60" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs062.html#___sec61" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs063.html#___sec62" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs064.html#___sec63" style="font-size: 80%;">Simple implementation of the Conjugate gradient algorithm</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs065.html#___sec64" style="font-size: 80%;">BroydenFletcherGoldfarbShanno algorithm</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs066.html#___sec65" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs067.html#___sec66" style="font-size: 80%;">Momentum based GD</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs068.html#___sec67" style="font-size: 80%;">More on momentum based approaches</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs069.html#___sec68" style="font-size: 80%;">Momentum parameter</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs070.html#___sec69" style="font-size: 80%;">Second moment of the gradient</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs071.html#___sec70" style="font-size: 80%;">RMS prop</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs072.html#___sec71" style="font-size: 80%;">ADAM optimizer</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs073.html#___sec72" style="font-size: 80%;">Practical tips</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs027.html#___sec26" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs028.html#___sec27" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs029.html#___sec28" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs030.html#___sec29" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs031.html#___sec30" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs032.html#___sec31" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs033.html#___sec32" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs034.html#___sec33" style="font-size: 80%;">Conjugate gradient method</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs035.html#___sec34" style="font-size: 80%;">Simple implementation of the Conjugate gradient algorithm</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs036.html#___sec35" style="font-size: 80%;">BroydenFletcherGoldfarbShanno algorithm</a></li>
<!-- navigation toc: --> <li><a href="#___sec36" style="font-size: 80%;">Revisiting our first homework</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs038.html#___sec37" style="font-size: 80%;">Gradient descent example</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs039.html#___sec38" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs040.html#___sec39" style="font-size: 80%;">The Hessian matrix</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs041.html#___sec40" style="font-size: 80%;">Simple program</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs042.html#___sec41" style="font-size: 80%;">Gradient Descent Example</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs043.html#___sec42" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs044.html#___sec43" style="font-size: 80%;">Gradient descent and Ridge</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs045.html#___sec44" style="font-size: 80%;">Automatic differentiation</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs046.html#___sec45" style="font-size: 80%;">Using autograd</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs047.html#___sec46" style="font-size: 80%;">Autograd with more complicated functions</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs048.html#___sec47" style="font-size: 80%;">More complicated functions using the elements of their arguments directly</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs049.html#___sec48" style="font-size: 80%;">Functions using mathematical functions from Numpy</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs050.html#___sec49" style="font-size: 80%;">More autograd</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs051.html#___sec50" style="font-size: 80%;">And with loops</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs052.html#___sec51" style="font-size: 80%;">Using recursion</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs053.html#___sec52" style="font-size: 80%;">Unsupported functions</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs054.html#___sec53" style="font-size: 80%;">The syntax a.dot(b) when finding the dot product</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs055.html#___sec54" style="font-size: 80%;">Recommended to avoid</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs056.html#___sec55" style="font-size: 80%;">Stochastic Gradient Descent</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs057.html#___sec56" style="font-size: 80%;">Computation of gradients</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs058.html#___sec57" style="font-size: 80%;">SGD example</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs059.html#___sec58" style="font-size: 80%;">The gradient step</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs060.html#___sec59" style="font-size: 80%;">Simple example code</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs061.html#___sec60" style="font-size: 80%;">When do we stop?</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs062.html#___sec61" style="font-size: 80%;">Slightly different approach</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs063.html#___sec62" style="font-size: 80%;">Program for stochastic gradient</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs064.html#___sec63" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs065.html#___sec64" style="font-size: 80%;">Momentum based GD</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs066.html#___sec65" style="font-size: 80%;">More on momentum based approaches</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs067.html#___sec66" style="font-size: 80%;">Momentum parameter</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs068.html#___sec67" style="font-size: 80%;">Second moment of the gradient</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs069.html#___sec68" style="font-size: 80%;">RMS prop</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs070.html#___sec69" style="font-size: 80%;">ADAM optimizer</a></li>
<!-- navigation toc: --> <li><a href="._Splines-bs071.html#___sec70" style="font-size: 80%;">Practical tips</a></li>
</ul>
</li>
@@ -274,36 +270,36 @@ MathJax.Hub.Config({
<a name="part0037"></a>
<!-- !split -->
<h2 id="___sec36" class="anchor">Using autograd </h2>
<h2 id="___sec36" class="anchor">Revisiting our first homework </h2>
<p>
Here we
experiment with what kind of functions Autograd is capable
of finding the gradient of. The following Python functions are just
meant to illustrate what Autograd can do, but please feel free to
experiment with other, possibly more complicated, functions as well.
We will use linear regression as a case study for the gradient descent
methods. Linear regression is a great test case for the gradient
descent methods discussed in the lectures since it has several
desirable properties such as:
<p>
<ol>
<li> An analytical solution (recall homework set 1).</li>
<li> The gradient can be computed analytically.</li>
<li> The cost function is convex which guarantees that gradient descent converges for small enough learning rates</li>
</ol>
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
We revisit the example from homework set 1 where we had
$$
y_i = 5x_i^2 + 0.1\xi_i, \ i=1,\cdots,100
$$
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f1</span>(x):
<span style="color: #008000; font-weight: bold">return</span> x<span style="color: #666666">**3</span> <span style="color: #666666">+</span> <span style="color: #666666">1</span>
with \( x_i \in [0,1] \) chosen randomly with a uniform distribution. Additionally \( \xi_i \) represents stochastic noise chosen according to a normal distribution \( \cal {N}(0,1) \).
The linear regression model is given by
$$
h_\beta(x) = \hat{y} = \beta_0 + \beta_1 x,
$$
f1_grad <span style="color: #666666">=</span> grad(f1)
such that
$$
\hat{y}_i = \beta_0 + \beta_1 x_i.
$$
<span style="color: #408080; font-style: italic"># Remember to send in float as argument to the computed gradient from Autograd!</span>
a <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
<span style="color: #408080; font-style: italic"># See the evaluated gradient at a using autograd:</span>
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">&quot;The gradient of f1 evaluated at a = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> using autograd is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">&quot;</span><span style="color: #666666">%</span>(a,f1_grad(a)))
<span style="color: #408080; font-style: italic"># Compare with the analytical derivative, that is f1&#39;(x) = 3*x**2 </span>
grad_analytical <span style="color: #666666">=</span> <span style="color: #666666">3*</span>a<span style="color: #666666">**2</span>
<span style="color: #008000; font-weight: bold">print</span>(<span style="color: #BA2121">&quot;The gradient of f1 evaluated at a = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> by finding the analytic expression is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">&quot;</span><span style="color: #666666">%</span>(a,grad_analytical))
</pre></div>
<p>
<p>
<!-- navigation buttons at the bottom of the page -->
@@ -330,7 +326,7 @@ grad_analytical <span style="color: #666666">=</span> <span style="color: #66666
<li><a href="._Splines-bs045.html">46</a></li>
<li><a href="._Splines-bs046.html">47</a></li>
<li><a href="">...</a></li>
<li><a href="._Splines-bs073.html">74</a></li>
<li><a href="._Splines-bs071.html">72</a></li>
<li><a href="._Splines-bs038.html">&raquo;</a></li>
</ul>
<!-- ------------------- end of main content --------------- -->