Files
FYS-STK4155/doc/pub/week40/html/._week40-bs009.html
T
2020-09-29 17:02:23 +02:00

334 lines
21 KiB
HTML

<!--
Automatically generated HTML file from DocOnce source
(https://github.com/hplgit/doconce/)
-->
<html>
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<meta name="description" content="Week 40: From Stochastic Gradient Descent to Neural networks">
<title>Week 40: From Stochastic Gradient Descent to Neural networks</title>
<!-- Bootstrap style: bootstrap -->
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
<!-- not necessary
<link href="https://netdna.bootstrapcdn.com/font-awesome/4.0.3/css/font-awesome.css" rel="stylesheet">
-->
<style type="text/css">
/* Add scrollbar to dropdown menus in bootstrap navigation bar */
.dropdown-menu {
height: auto;
max-height: 400px;
overflow-x: hidden;
}
/* Adds an invisible element before each target to offset for the navigation
bar */
.anchor::before {
content:"";
display:block;
height:50px; /* fixed header height for style bootstrap */
margin:-50px 0 0; /* negative fixed header height */
}
</style>
</head>
<!-- tocinfo
{'highest level': 2,
'sections': [('Plan for week 40', 2, None, '___sec0'),
('Overview video for week 40', 2, None, '___sec1'),
('Stochastic Gradient Descent', 2, None, '___sec2'),
('Computation of gradients', 2, None, '___sec3'),
('SGD example', 2, None, '___sec4'),
('The gradient step', 2, None, '___sec5'),
('Simple example code', 2, None, '___sec6'),
('When do we stop?', 2, None, '___sec7'),
('Slightly different approach', 2, None, '___sec8'),
('Program for stochastic gradient', 2, None, '___sec9'),
('Momentum based GD', 2, None, '___sec10'),
('More on momentum based approaches', 2, None, '___sec11'),
('Momentum parameter', 2, None, '___sec12'),
('Second moment of the gradient', 2, None, '___sec13'),
('RMS prop', 2, None, '___sec14'),
('ADAM optimizer', 2, None, '___sec15'),
('Practical tips', 2, None, '___sec16'),
('Automatic differentiation', 2, None, '___sec17'),
('Using autograd', 2, None, '___sec18'),
('Autograd with more complicated functions', 2, None, '___sec19'),
('More complicated functions using the elements of their '
'arguments directly',
2,
None,
'___sec20'),
('Functions using mathematical functions from Numpy',
2,
None,
'___sec21'),
('More autograd', 2, None, '___sec22'),
('And with loops', 2, None, '___sec23'),
('Using recursion', 2, None, '___sec24'),
('Unsupported functions', 2, None, '___sec25'),
('The syntax a.dot(b) when finding the dot product',
2,
None,
'___sec26'),
('Recommended to avoid', 2, None, '___sec27'),
('Neural networks', 2, None, '___sec28'),
('Artificial neurons', 2, None, '___sec29'),
('Neural network types', 2, None, '___sec30'),
('Feed-forward neural networks', 2, None, '___sec31'),
('Convolutional Neural Network', 2, None, '___sec32'),
('Recurrent neural networks', 2, None, '___sec33'),
('Other types of networks', 2, None, '___sec34'),
('Multilayer perceptrons', 2, None, '___sec35'),
('Why multilayer perceptrons?', 2, None, '___sec36'),
('Mathematical model', 2, None, '___sec37'),
('Mathematical model', 2, None, '___sec38'),
('Mathematical model', 2, None, '___sec39'),
('Mathematical model', 2, None, '___sec40'),
('Mathematical model', 2, None, '___sec41'),
('Matrix-vector notation', 3, None, '___sec42'),
('Matrix-vector notation and activation', 3, None, '___sec43'),
('Activation functions', 3, None, '___sec44'),
('Activation functions, Logistic and Hyperbolic ones',
3,
None,
'___sec45'),
('Relevance', 3, None, '___sec46'),
('The multilayer perceptron (MLP)', 2, None, '___sec47'),
('From one to many layers, the universal approximation theorem',
2,
None,
'___sec48'),
('Deriving the back propagation code for a multilayer perceptron '
'model',
2,
None,
'___sec49'),
('Definitions', 2, None, '___sec50'),
('Derivatives and the chain rule', 2, None, '___sec51'),
('Derivative of the cost function', 2, None, '___sec52'),
('Bringing it together, first back propagation equation',
2,
None,
'___sec53'),
('Derivatives in terms of $z_j^L$', 2, None, '___sec54'),
('Bringing it together', 2, None, '___sec55'),
('Final back propagating equation', 2, None, '___sec56'),
('Setting up the Back propagation algorithm',
2,
None,
'___sec57')]}
end of tocinfo -->
<body>
<script type="text/x-mathjax-config">
MathJax.Hub.Config({
TeX: {
equationNumbers: { autoNumber: "none" },
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
}
});
</script>
<script type="text/javascript" async
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
</script>
<!-- Bootstrap navigation bar -->
<div class="navbar navbar-default navbar-fixed-top">
<div class="navbar-header">
<button type="button" class="navbar-toggle" data-toggle="collapse" data-target=".navbar-responsive-collapse">
<span class="icon-bar"></span>
<span class="icon-bar"></span>
<span class="icon-bar"></span>
</button>
<a class="navbar-brand" href="week40-bs.html">Week 40: From Stochastic Gradient Descent to Neural networks</a>
</div>
<div class="navbar-collapse collapse navbar-responsive-collapse">
<ul class="nav navbar-nav navbar-right">
<li class="dropdown">
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
<ul class="dropdown-menu">
<!-- navigation toc: --> <li><a href="._week40-bs001.html#___sec0" style="font-size: 80%;"><b>Plan for week 40</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs002.html#___sec1" style="font-size: 80%;"><b>Overview video for week 40</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs003.html#___sec2" style="font-size: 80%;"><b>Stochastic Gradient Descent</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs004.html#___sec3" style="font-size: 80%;"><b>Computation of gradients</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs005.html#___sec4" style="font-size: 80%;"><b>SGD example</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs006.html#___sec5" style="font-size: 80%;"><b>The gradient step</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs007.html#___sec6" style="font-size: 80%;"><b>Simple example code</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs008.html#___sec7" style="font-size: 80%;"><b>When do we stop?</b></a></li>
<!-- navigation toc: --> <li><a href="#___sec8" style="font-size: 80%;"><b>Slightly different approach</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs010.html#___sec9" style="font-size: 80%;"><b>Program for stochastic gradient</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs011.html#___sec10" style="font-size: 80%;"><b>Momentum based GD</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs012.html#___sec11" style="font-size: 80%;"><b>More on momentum based approaches</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs013.html#___sec12" style="font-size: 80%;"><b>Momentum parameter</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs014.html#___sec13" style="font-size: 80%;"><b>Second moment of the gradient</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs015.html#___sec14" style="font-size: 80%;"><b>RMS prop</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs016.html#___sec15" style="font-size: 80%;"><b>ADAM optimizer</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs017.html#___sec16" style="font-size: 80%;"><b>Practical tips</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs018.html#___sec17" style="font-size: 80%;"><b>Automatic differentiation</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs019.html#___sec18" style="font-size: 80%;"><b>Using autograd</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs020.html#___sec19" style="font-size: 80%;"><b>Autograd with more complicated functions</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs021.html#___sec20" style="font-size: 80%;"><b>More complicated functions using the elements of their arguments directly</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs022.html#___sec21" style="font-size: 80%;"><b>Functions using mathematical functions from Numpy</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs023.html#___sec22" style="font-size: 80%;"><b>More autograd</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs024.html#___sec23" style="font-size: 80%;"><b>And with loops</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs025.html#___sec24" style="font-size: 80%;"><b>Using recursion</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs026.html#___sec25" style="font-size: 80%;"><b>Unsupported functions</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs027.html#___sec26" style="font-size: 80%;"><b>The syntax a.dot(b) when finding the dot product</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs028.html#___sec27" style="font-size: 80%;"><b>Recommended to avoid</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs029.html#___sec28" style="font-size: 80%;"><b>Neural networks</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs030.html#___sec29" style="font-size: 80%;"><b>Artificial neurons</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs031.html#___sec30" style="font-size: 80%;"><b>Neural network types</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs032.html#___sec31" style="font-size: 80%;"><b>Feed-forward neural networks</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs033.html#___sec32" style="font-size: 80%;"><b>Convolutional Neural Network</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs034.html#___sec33" style="font-size: 80%;"><b>Recurrent neural networks</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs035.html#___sec34" style="font-size: 80%;"><b>Other types of networks</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs036.html#___sec35" style="font-size: 80%;"><b>Multilayer perceptrons</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs037.html#___sec36" style="font-size: 80%;"><b>Why multilayer perceptrons?</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs038.html#___sec37" style="font-size: 80%;"><b>Mathematical model</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs039.html#___sec38" style="font-size: 80%;"><b>Mathematical model</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs040.html#___sec39" style="font-size: 80%;"><b>Mathematical model</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs041.html#___sec40" style="font-size: 80%;"><b>Mathematical model</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs042.html#___sec41" style="font-size: 80%;"><b>Mathematical model</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs043.html#___sec42" style="font-size: 80%;">&nbsp;&nbsp;&nbsp;Matrix-vector notation</a></li>
<!-- navigation toc: --> <li><a href="._week40-bs044.html#___sec43" style="font-size: 80%;">&nbsp;&nbsp;&nbsp;Matrix-vector notation and activation</a></li>
<!-- navigation toc: --> <li><a href="._week40-bs045.html#___sec44" style="font-size: 80%;">&nbsp;&nbsp;&nbsp;Activation functions</a></li>
<!-- navigation toc: --> <li><a href="._week40-bs046.html#___sec45" style="font-size: 80%;">&nbsp;&nbsp;&nbsp;Activation functions, Logistic and Hyperbolic ones</a></li>
<!-- navigation toc: --> <li><a href="._week40-bs047.html#___sec46" style="font-size: 80%;">&nbsp;&nbsp;&nbsp;Relevance</a></li>
<!-- navigation toc: --> <li><a href="._week40-bs048.html#___sec47" style="font-size: 80%;"><b>The multilayer perceptron (MLP)</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs049.html#___sec48" style="font-size: 80%;"><b>From one to many layers, the universal approximation theorem</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs050.html#___sec49" style="font-size: 80%;"><b>Deriving the back propagation code for a multilayer perceptron model</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs051.html#___sec50" style="font-size: 80%;"><b>Definitions</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs052.html#___sec51" style="font-size: 80%;"><b>Derivatives and the chain rule</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs053.html#___sec52" style="font-size: 80%;"><b>Derivative of the cost function</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs054.html#___sec53" style="font-size: 80%;"><b>Bringing it together, first back propagation equation</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs055.html#___sec54" style="font-size: 80%;"><b>Derivatives in terms of \( z_j^L \)</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs056.html#___sec55" style="font-size: 80%;"><b>Bringing it together</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs057.html#___sec56" style="font-size: 80%;"><b>Final back propagating equation</b></a></li>
<!-- navigation toc: --> <li><a href="._week40-bs058.html#___sec57" style="font-size: 80%;"><b>Setting up the Back propagation algorithm</b></a></li>
</ul>
</li>
</ul>
</div>
</div>
</div> <!-- end of navigation bar -->
<div class="container">
<p>&nbsp;</p><p>&nbsp;</p><p>&nbsp;</p> <!-- add vertical space -->
<a name="part0009"></a>
<!-- !split -->
<h2 id="___sec8" class="anchor">Slightly different approach </h2>
<p>
Another approach is to let the step length \( \gamma_j \) depend on the
number of epochs in such a way that it becomes very small after a
reasonable time such that we do not move at all.
<p>
As an example, let \( e = 0,1,2,3,\cdots \) denote the current epoch and let \( t_0, t_1 > 0 \) be two fixed numbers. Furthermore, let \( t = e \cdot m + i \) where \( m \) is the number of minibatches and \( i=0,\cdots,m-1 \). Then the function $$\gamma_j(t; t_0, t_1) = \frac{t_0}{t+t_1} $$ goes to zero as the number of epochs gets large. I.e. we start with a step length \( \gamma_j (0; t_0, t_1) = t_0/t_1 \) which decays in <em>time</em> \( t \).
<p>
In this way we can fix the number of epochs, compute \( \beta \) and
evaluate the cost function at the end. Repeating the computation will
give a different result since the scheme is random by design. Then we
pick the final \( \beta \) that gives the lowest value of the cost
function.
<p>
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">step_length</span>(t,t0,t1):
<span style="color: #008000; font-weight: bold">return</span> t0<span style="color: #666666">/</span>(t<span style="color: #666666">+</span>t1)
n <span style="color: #666666">=</span> <span style="color: #666666">100</span> <span style="color: #408080; font-style: italic">#100 datapoints </span>
M <span style="color: #666666">=</span> <span style="color: #666666">5</span> <span style="color: #408080; font-style: italic">#size of each minibatch</span>
m <span style="color: #666666">=</span> <span style="color: #008000">int</span>(n<span style="color: #666666">/</span>M) <span style="color: #408080; font-style: italic">#number of minibatches</span>
n_epochs <span style="color: #666666">=</span> <span style="color: #666666">500</span> <span style="color: #408080; font-style: italic">#number of epochs</span>
t0 <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
t1 <span style="color: #666666">=</span> <span style="color: #666666">10</span>
gamma_j <span style="color: #666666">=</span> t0<span style="color: #666666">/</span>t1
j <span style="color: #666666">=</span> <span style="color: #666666">0</span>
<span style="color: #008000; font-weight: bold">for</span> epoch <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,n_epochs<span style="color: #666666">+1</span>):
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(m):
k <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randint(m) <span style="color: #408080; font-style: italic">#Pick the k-th minibatch at random</span>
<span style="color: #408080; font-style: italic">#Compute the gradient using the data in minibatch Bk</span>
<span style="color: #408080; font-style: italic">#Compute new suggestion for beta</span>
t <span style="color: #666666">=</span> epoch<span style="color: #666666">*</span>m<span style="color: #666666">+</span>i
gamma_j <span style="color: #666666">=</span> step_length(t,t0,t1)
j <span style="color: #666666">+=</span> <span style="color: #666666">1</span>
<span style="color: #008000">print</span>(<span style="color: #BA2121">&quot;gamma_j after </span><span style="color: #BB6688; font-weight: bold">%d</span><span style="color: #BA2121"> epochs: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">&quot;</span> <span style="color: #666666">%</span> (n_epochs,gamma_j))
</pre></div>
<p>
<p>
<!-- navigation buttons at the bottom of the page -->
<ul class="pagination">
<li><a href="._week40-bs008.html">&laquo;</a></li>
<li><a href="._week40-bs000.html">1</a></li>
<li><a href="._week40-bs001.html">2</a></li>
<li><a href="._week40-bs002.html">3</a></li>
<li><a href="._week40-bs003.html">4</a></li>
<li><a href="._week40-bs004.html">5</a></li>
<li><a href="._week40-bs005.html">6</a></li>
<li><a href="._week40-bs006.html">7</a></li>
<li><a href="._week40-bs007.html">8</a></li>
<li><a href="._week40-bs008.html">9</a></li>
<li class="active"><a href="._week40-bs009.html">10</a></li>
<li><a href="._week40-bs010.html">11</a></li>
<li><a href="._week40-bs011.html">12</a></li>
<li><a href="._week40-bs012.html">13</a></li>
<li><a href="._week40-bs013.html">14</a></li>
<li><a href="._week40-bs014.html">15</a></li>
<li><a href="._week40-bs015.html">16</a></li>
<li><a href="._week40-bs016.html">17</a></li>
<li><a href="._week40-bs017.html">18</a></li>
<li><a href="._week40-bs018.html">19</a></li>
<li><a href="">...</a></li>
<li><a href="._week40-bs058.html">59</a></li>
<li><a href="._week40-bs010.html">&raquo;</a></li>
</ul>
<!-- ------------------- end of main content --------------- -->
</div> <!-- end container -->
<!-- include javascript, jQuery *first* -->
<script src="https://ajax.googleapis.com/ajax/libs/jquery/1.10.2/jquery.min.js"></script>
<script src="https://netdna.bootstrapcdn.com/bootstrap/3.0.0/js/bootstrap.min.js"></script>
<!-- Bootstrap footer
<footer>
<a href="http://..."><img width="250" align=right src="http://..."></a>
</footer>
-->
<center style="font-size:80%">
<!-- copyright only on the titlepage -->
</center>
</body>
</html>