2335 lines
158 KiB
HTML
2335 lines
158 KiB
HTML
<!--
|
|
HTML file automatically generated from DocOnce source
|
|
(https://github.com/doconce/doconce/)
|
|
doconce format html week40.do.txt --pygments_html_style=default --html_style=bloodish --html_links_in_new_window --html_output=week40 --no_mako
|
|
-->
|
|
<html>
|
|
<head>
|
|
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
|
<meta name="generator" content="DocOnce: https://github.com/doconce/doconce/" />
|
|
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
<meta name="description" content="Week 40: Gradient descent methods (continued) and start Neural networks">
|
|
<title>Week 40: Gradient descent methods (continued) and start Neural networks</title>
|
|
<style type="text/css">
|
|
/* bloodish style */
|
|
body {
|
|
font-family: Helvetica, Verdana, Arial, Sans-serif;
|
|
color: #404040;
|
|
background: #ffffff;
|
|
}
|
|
h1 { font-size: 1.8em; color: #8A0808; }
|
|
h2 { font-size: 1.6em; color: #8A0808; }
|
|
h3 { font-size: 1.4em; color: #8A0808; }
|
|
h4 { font-size: 1.2em; color: #8A0808; }
|
|
a { color: #8A0808; text-decoration:none; }
|
|
tt { font-family: "Courier New", Courier; }
|
|
p { text-indent: 0px; }
|
|
hr { border: 0; width: 80%; border-bottom: 1px solid #aaa}
|
|
p.caption { width: 80%; font-style: normal; text-align: left; }
|
|
hr.figure { border: 0; width: 80%; border-bottom: 1px solid #aaa; }div.highlight {
|
|
border: 1px solid #cfcfcf;
|
|
border-radius: 2px;
|
|
line-height: 1.21429em;
|
|
}
|
|
div.cell {
|
|
width: 100%;
|
|
padding: 5px 5px 5px 0;
|
|
margin: 0;
|
|
outline: none;
|
|
}
|
|
div.input {
|
|
page-break-inside: avoid;
|
|
box-orient: horizontal;
|
|
box-align: stretch;
|
|
display: flex;
|
|
flex-direction: row;
|
|
align-items: stretch;
|
|
}
|
|
div.inner_cell {
|
|
box-orient: vertical;
|
|
box-align: stretch;
|
|
display: flex;
|
|
flex-direction: column;
|
|
align-items: stretch;
|
|
box-flex: 1;
|
|
flex: 1;
|
|
}
|
|
div.input_area {
|
|
border: 1px solid #cfcfcf;
|
|
border-radius: 4px;
|
|
background: #f7f7f7;
|
|
line-height: 1.21429em;
|
|
}
|
|
div.input_area > div.highlight {
|
|
margin: .4em;
|
|
border: none;
|
|
padding: 0;
|
|
background-color: transparent;
|
|
}
|
|
div.output_wrapper {
|
|
position: relative;
|
|
box-orient: vertical;
|
|
box-align: stretch;
|
|
display: flex;
|
|
flex-direction: column;
|
|
align-items: stretch;
|
|
}
|
|
.output {
|
|
box-orient: vertical;
|
|
box-align: stretch;
|
|
display: flex;
|
|
flex-direction: column;
|
|
align-items: stretch;
|
|
}
|
|
div.output_area {
|
|
padding: 0;
|
|
page-break-inside: avoid;
|
|
box-orient: horizontal;
|
|
box-align: stretch;
|
|
display: flex;
|
|
flex-direction: row;
|
|
align-items: stretch;
|
|
}
|
|
div.output_subarea {
|
|
padding: .4em .4em 0 .4em;
|
|
box-flex: 1;
|
|
flex: 1;
|
|
}
|
|
div.output_text {
|
|
text-align: left;
|
|
color: #000;
|
|
line-height: 1.21429em;
|
|
}
|
|
.alert-text-small { font-size: 80%; }
|
|
.alert-text-large { font-size: 130%; }
|
|
.alert-text-normal { font-size: 90%; }
|
|
.alert {
|
|
padding:8px 35px 8px 14px; margin-bottom:18px;
|
|
text-shadow:0 1px 0 rgba(255,255,255,0.5);
|
|
border:1px solid #bababa;
|
|
border-radius: 4px;
|
|
-webkit-border-radius: 4px;
|
|
-moz-border-radius: 4px;
|
|
color: #555;
|
|
background-color: #f8f8f8;
|
|
background-position: 10px 5px;
|
|
background-repeat: no-repeat;
|
|
background-size: 38px;
|
|
padding-left: 55px;
|
|
width: 75%;
|
|
}
|
|
.alert-block {padding-top:14px; padding-bottom:14px}
|
|
.alert-block > p, .alert-block > ul {margin-bottom:1em}
|
|
.alert li {margin-top: 1em}
|
|
.alert-block p+p {margin-top:5px}
|
|
.alert-notice { background-image: url(https://cdn.rawgit.com/doconce/doconce/master/bundled/html_images/small_gray_notice.png); }
|
|
.alert-summary { background-image:url(https://cdn.rawgit.com/doconce/doconce/master/bundled/html_images/small_gray_summary.png); }
|
|
.alert-warning { background-image: url(https://cdn.rawgit.com/doconce/doconce/master/bundled/html_images/small_gray_warning.png); }
|
|
.alert-question {background-image:url(https://cdn.rawgit.com/doconce/doconce/master/bundled/html_images/small_gray_question.png); }
|
|
div { text-align: justify; text-justify: inter-word; }
|
|
.tab {
|
|
padding-left: 1.5em;
|
|
}
|
|
div.toc p,a {
|
|
line-height: 1.3;
|
|
margin-top: 1.1;
|
|
margin-bottom: 1.1;
|
|
}
|
|
</style>
|
|
</head>
|
|
|
|
<!-- tocinfo
|
|
{'highest level': 2,
|
|
'sections': [('Lecture Monday September 30, 2024',
|
|
2,
|
|
None,
|
|
'lecture-monday-september-30-2024'),
|
|
('Suggested readings and videos',
|
|
2,
|
|
None,
|
|
'suggested-readings-and-videos'),
|
|
('Lab sessions Tuesday and Wednesday',
|
|
2,
|
|
None,
|
|
'lab-sessions-tuesday-and-wednesday'),
|
|
('Automatic differentiation',
|
|
2,
|
|
None,
|
|
'automatic-differentiation'),
|
|
('Using autograd', 2, None, 'using-autograd'),
|
|
('Autograd with more complicated functions',
|
|
2,
|
|
None,
|
|
'autograd-with-more-complicated-functions'),
|
|
('More complicated functions using the elements of their '
|
|
'arguments directly',
|
|
2,
|
|
None,
|
|
'more-complicated-functions-using-the-elements-of-their-arguments-directly'),
|
|
('Functions using mathematical functions from Numpy',
|
|
2,
|
|
None,
|
|
'functions-using-mathematical-functions-from-numpy'),
|
|
('More autograd', 2, None, 'more-autograd'),
|
|
('And with loops', 2, None, 'and-with-loops'),
|
|
('Using recursion', 2, None, 'using-recursion'),
|
|
('Using Autograd with OLS', 2, None, 'using-autograd-with-ols'),
|
|
('Same code but now with momentum gradient descent',
|
|
2,
|
|
None,
|
|
'same-code-but-now-with-momentum-gradient-descent'),
|
|
('Including Stochastic Gradient Descent with Autograd',
|
|
2,
|
|
None,
|
|
'including-stochastic-gradient-descent-with-autograd'),
|
|
('Same code but now with momentum gradient descent',
|
|
2,
|
|
None,
|
|
'same-code-but-now-with-momentum-gradient-descent'),
|
|
('Similar (second order function now) problem but now with '
|
|
'AdaGrad',
|
|
2,
|
|
None,
|
|
'similar-second-order-function-now-problem-but-now-with-adagrad'),
|
|
('RMSprop for adaptive learning rate with Stochastic Gradient '
|
|
'Descent',
|
|
2,
|
|
None,
|
|
'rmsprop-for-adaptive-learning-rate-with-stochastic-gradient-descent'),
|
|
('And finally "ADAM":"https://arxiv.org/pdf/1412.6980.pdf"',
|
|
2,
|
|
None,
|
|
'and-finally-adam-https-arxiv-org-pdf-1412-6980-pdf'),
|
|
('And Logistic Regression', 2, None, 'and-logistic-regression'),
|
|
('Introducing "JAX":"https://jax.readthedocs.io/en/latest/"',
|
|
2,
|
|
None,
|
|
'introducing-jax-https-jax-readthedocs-io-en-latest'),
|
|
('Getting started with Jax, note the way we import numpy',
|
|
3,
|
|
None,
|
|
'getting-started-with-jax-note-the-way-we-import-numpy'),
|
|
('A warm-up example', 3, None, 'a-warm-up-example'),
|
|
('A more advanced example', 3, None, 'a-more-advanced-example'),
|
|
('Introduction to Neural networks',
|
|
2,
|
|
None,
|
|
'introduction-to-neural-networks'),
|
|
('Artificial neurons', 2, None, 'artificial-neurons'),
|
|
('Neural network types', 2, None, 'neural-network-types'),
|
|
('Feed-forward neural networks',
|
|
2,
|
|
None,
|
|
'feed-forward-neural-networks'),
|
|
('Convolutional Neural Network',
|
|
2,
|
|
None,
|
|
'convolutional-neural-network'),
|
|
('Recurrent neural networks',
|
|
2,
|
|
None,
|
|
'recurrent-neural-networks'),
|
|
('Other types of networks', 2, None, 'other-types-of-networks'),
|
|
('Multilayer perceptrons', 2, None, 'multilayer-perceptrons'),
|
|
('Why multilayer perceptrons?',
|
|
2,
|
|
None,
|
|
'why-multilayer-perceptrons'),
|
|
('Illustration of a single perceptron model and a '
|
|
'multi-perceptron model',
|
|
2,
|
|
None,
|
|
'illustration-of-a-single-perceptron-model-and-a-multi-perceptron-model'),
|
|
('Examples of XOR, OR and AND gates',
|
|
2,
|
|
None,
|
|
'examples-of-xor-or-and-and-gates'),
|
|
('Does Logistic Regression do a better Job?',
|
|
2,
|
|
None,
|
|
'does-logistic-regression-do-a-better-job'),
|
|
('Adding Neural Networks', 2, None, 'adding-neural-networks'),
|
|
('Mathematical model', 2, None, 'mathematical-model'),
|
|
('Mathematical model', 2, None, 'mathematical-model'),
|
|
('Mathematical model', 2, None, 'mathematical-model'),
|
|
('Mathematical model', 2, None, 'mathematical-model'),
|
|
('Mathematical model', 2, None, 'mathematical-model'),
|
|
('Matrix-vector notation', 3, None, 'matrix-vector-notation'),
|
|
('Matrix-vector notation and activation',
|
|
3,
|
|
None,
|
|
'matrix-vector-notation-and-activation'),
|
|
('Activation functions', 3, None, 'activation-functions'),
|
|
('Activation functions, Logistic and Hyperbolic ones',
|
|
3,
|
|
None,
|
|
'activation-functions-logistic-and-hyperbolic-ones'),
|
|
('Relevance', 3, None, 'relevance')]}
|
|
end of tocinfo -->
|
|
|
|
<body>
|
|
|
|
|
|
|
|
<script type="text/x-mathjax-config">
|
|
MathJax.Hub.Config({
|
|
TeX: {
|
|
equationNumbers: { autoNumber: "AMS" },
|
|
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
|
|
}
|
|
});
|
|
</script>
|
|
<script type="text/javascript" async
|
|
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
|
|
</script>
|
|
|
|
|
|
<!-- ------------------- main content ---------------------- -->
|
|
<center>
|
|
<h1>Week 40: Gradient descent methods (continued) and start Neural networks</h1>
|
|
</center> <!-- document title -->
|
|
|
|
<!-- author(s): Morten Hjorth-Jensen -->
|
|
<center>
|
|
<b>Morten Hjorth-Jensen</b>
|
|
</center>
|
|
<!-- institution -->
|
|
<center>
|
|
<b>Department of Physics, University of Oslo, Norway</b>
|
|
</center>
|
|
<br>
|
|
<center>
|
|
<h4>September 29-October 3, 2025</h4>
|
|
</center> <!-- date -->
|
|
<br>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="lecture-monday-september-30-2024">Lecture Monday September 30, 2024 </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b></b>
|
|
<p>
|
|
<ol>
|
|
<li> Stochastic Gradient descent with examples and automatic differentiation</li>
|
|
<li> If we get time, we start with the basics of Neural Networks, setting up the basic steps, from the simple perceptron model to the multi-layer perceptron model</li>
|
|
<li> <a href="https://youtu.be/jdJoOrCIdII" target="_blank">Video of lecture</a></li>
|
|
<li> Whiteboard notes at <a href="https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/2024/NotesSeptember30.pdf" target="_blank"><tt>https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/2024/NotesSeptember30.pdf</tt></a></li>
|
|
</ol>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="suggested-readings-and-videos">Suggested readings and videos </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b>Readings and Videos:</b>
|
|
<p>
|
|
<ol>
|
|
<li> The lecture notes for week 40 (these notes)</li>
|
|
<li> For a good discussion on gradient methods, we would like to recommend Goodfellow et al section 4.3-4.5 and sections 8.3-8.6. We will come back to the latter chapter in our discussion of Neural networks as well.</li>
|
|
<li> For neural networks we recommend Goodfellow et al chapter 6 and Raschka et al chapter 2 (contains also material about gradient descent) and chapter 11 (we will use this next week)</li>
|
|
<li> Video on gradient descent at <a href="https://www.youtube.com/watch?v=sDv4f4s2SB8" target="_blank"><tt>https://www.youtube.com/watch?v=sDv4f4s2SB8</tt></a></li>
|
|
<li> Video on stochastic gradient descent at <a href="https://www.youtube.com/watch?v=vMh0zPT0tLI" target="_blank"><tt>https://www.youtube.com/watch?v=vMh0zPT0tLI</tt></a></li>
|
|
<li> Neural Networks demystified at <a href="https://www.youtube.com/watch?v=bxe2T-V8XRs&list=PLiaHhY2iBX9hdHaRr6b7XevZtgZRa1PoU&ab_channel=WelchLabs" target="_blank"><tt>https://www.youtube.com/watch?v=bxe2T-V8XRs&list=PLiaHhY2iBX9hdHaRr6b7XevZtgZRa1PoU&ab_channel=WelchLabs</tt></a></li>
|
|
<li> Building Neural Networks from scratch at URL:https://www.youtube.com/watch?v=Wo5dMEP_BbI&list=PLQVvvaa0QuDcjD5BAw2DxE6OF2tius3V3&ab_channel=sentdex"</li>
|
|
</ol>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="lab-sessions-tuesday-and-wednesday">Lab sessions Tuesday and Wednesday </h2>
|
|
<div class="alert alert-block alert-block alert-text-normal">
|
|
<b>Material for the active learning sessions on Tuesday and Wednesday</b>
|
|
<p>
|
|
<ul>
|
|
<li> Work on project 1 and discussions on how to structure your report</li>
|
|
<li> No weekly exercises for week 40, project work only</li>
|
|
<li> Video on how to write scientific reports recorded during one of the lab sessions at <a href="https://youtu.be/tVW1ZDmZnwM" target="_blank"><tt>https://youtu.be/tVW1ZDmZnwM</tt></a></li>
|
|
<li> A general guideline can be found at <a href="https://github.com/CompPhysics/MachineLearning/blob/master/doc/Projects/EvaluationGrading/EvaluationForm.md" target="_blank"><tt>https://github.com/CompPhysics/MachineLearning/blob/master/doc/Projects/EvaluationGrading/EvaluationForm.md</tt></a>.</li>
|
|
</ul>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="automatic-differentiation">Automatic differentiation </h2>
|
|
|
|
<p><a href="https://en.wikipedia.org/wiki/Automatic_differentiation" target="_blank">Automatic differentiation (AD)</a>,
|
|
also called algorithmic
|
|
differentiation or computational differentiation,is a set of
|
|
techniques to numerically evaluate the derivative of a function
|
|
specified by a computer program. AD exploits the fact that every
|
|
computer program, no matter how complicated, executes a sequence of
|
|
elementary arithmetic operations (addition, subtraction,
|
|
multiplication, division, etc.) and elementary functions (exp, log,
|
|
sin, cos, etc.). By applying the chain rule repeatedly to these
|
|
operations, derivatives of arbitrary order can be computed
|
|
automatically, accurately to working precision, and using at most a
|
|
small constant factor more arithmetic operations than the original
|
|
program.
|
|
</p>
|
|
|
|
<p>Automatic differentiation is neither:</p>
|
|
|
|
<ul>
|
|
<li> Symbolic differentiation, nor</li>
|
|
<li> Numerical differentiation (the method of finite differences).</li>
|
|
</ul>
|
|
<p>Symbolic differentiation can lead to inefficient code and faces the
|
|
difficulty of converting a computer program into a single expression,
|
|
while numerical differentiation can introduce round-off errors in the
|
|
discretization process and cancellation
|
|
</p>
|
|
|
|
<p>Python has tools for so-called <b>automatic differentiation</b>.
|
|
Consider the following example
|
|
</p>
|
|
$$
|
|
f(x) = \sin\left(2\pi x + x^2\right)
|
|
$$
|
|
|
|
<p>which has the following derivative</p>
|
|
$$
|
|
f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right)
|
|
$$
|
|
|
|
<p>Using <b>autograd</b> we have</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># To do elementwise differentiation:</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> elementwise_grad <span style="color: #008000; font-weight: bold">as</span> egrad
|
|
|
|
<span style="color: #408080; font-style: italic"># To plot:</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f</span>(x):
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sin(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi<span style="color: #666666">*</span>x <span style="color: #666666">+</span> x<span style="color: #666666">**2</span>)
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f_grad_analytic</span>(x):
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>cos(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi<span style="color: #666666">*</span>x <span style="color: #666666">+</span> x<span style="color: #666666">**2</span>)<span style="color: #666666">*</span>(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi <span style="color: #666666">+</span> <span style="color: #666666">2*</span>x)
|
|
|
|
<span style="color: #408080; font-style: italic"># Do the comparison:</span>
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">0</span>,<span style="color: #666666">1</span>,<span style="color: #666666">1000</span>)
|
|
|
|
f_grad <span style="color: #666666">=</span> egrad(f)
|
|
|
|
computed <span style="color: #666666">=</span> f_grad(x)
|
|
analytic <span style="color: #666666">=</span> f_grad_analytic(x)
|
|
|
|
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">'Derivative computed from Autograd compared with the analytical derivative'</span>)
|
|
plt<span style="color: #666666">.</span>plot(x,computed,label<span style="color: #666666">=</span><span style="color: #BA2121">'autograd'</span>)
|
|
plt<span style="color: #666666">.</span>plot(x,analytic,label<span style="color: #666666">=</span><span style="color: #BA2121">'analytic'</span>)
|
|
|
|
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'x'</span>)
|
|
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'y'</span>)
|
|
plt<span style="color: #666666">.</span>legend()
|
|
|
|
plt<span style="color: #666666">.</span>show()
|
|
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The max absolute difference is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(np<span style="color: #666666">.</span>max(np<span style="color: #666666">.</span>abs(computed <span style="color: #666666">-</span> analytic))))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split -->
|
|
<h2 id="using-autograd">Using autograd </h2>
|
|
|
|
<p>Here we
|
|
experiment with what kind of functions Autograd is capable
|
|
of finding the gradient of. The following Python functions are just
|
|
meant to illustrate what Autograd can do, but please feel free to
|
|
experiment with other, possibly more complicated, functions as well.
|
|
</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f1</span>(x):
|
|
<span style="color: #008000; font-weight: bold">return</span> x<span style="color: #666666">**3</span> <span style="color: #666666">+</span> <span style="color: #666666">1</span>
|
|
|
|
f1_grad <span style="color: #666666">=</span> grad(f1)
|
|
|
|
<span style="color: #408080; font-style: italic"># Remember to send in float as argument to the computed gradient from Autograd!</span>
|
|
a <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># See the evaluated gradient at a using autograd:</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The gradient of f1 evaluated at a = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> using autograd is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(a,f1_grad(a)))
|
|
|
|
<span style="color: #408080; font-style: italic"># Compare with the analytical derivative, that is f1'(x) = 3*x**2 </span>
|
|
grad_analytical <span style="color: #666666">=</span> <span style="color: #666666">3*</span>a<span style="color: #666666">**2</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The gradient of f1 evaluated at a = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> by finding the analytic expression is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(a,grad_analytical))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="autograd-with-more-complicated-functions">Autograd with more complicated functions </h2>
|
|
|
|
<p>To differentiate with respect to two (or more) arguments of a Python
|
|
function, Autograd need to know at which variable the function if
|
|
being differentiated with respect to.
|
|
</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f2</span>(x1,x2):
|
|
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">3*</span>x1<span style="color: #666666">**3</span> <span style="color: #666666">+</span> x2<span style="color: #666666">*</span>(x1 <span style="color: #666666">-</span> <span style="color: #666666">5</span>) <span style="color: #666666">+</span> <span style="color: #666666">1</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1</span>
|
|
f2_grad_x1 <span style="color: #666666">=</span> grad(f2,<span style="color: #666666">0</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad</span>
|
|
f2_grad_x2 <span style="color: #666666">=</span> grad(f2,<span style="color: #666666">1</span>)
|
|
|
|
x1 <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
|
x2 <span style="color: #666666">=</span> <span style="color: #666666">3.0</span>
|
|
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Evaluating at x1 = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">, x2 = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x1,x2))
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"-"</span><span style="color: #666666">*30</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># Compare with the analytical derivatives:</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:</span>
|
|
f2_grad_x1_analytical <span style="color: #666666">=</span> <span style="color: #666666">9*</span>x1<span style="color: #666666">**2</span> <span style="color: #666666">+</span> x2
|
|
|
|
<span style="color: #408080; font-style: italic"># Derivative of f2 w.r.t x2 is: x1 - 5:</span>
|
|
f2_grad_x2_analytical <span style="color: #666666">=</span> x1 <span style="color: #666666">-</span> <span style="color: #666666">5</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># See the evaluated derivations:</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The derivative of f2 w.r.t x1: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>( f2_grad_x1(x1,x2) ))
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The analytical derivative of f2 w.r.t x1: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>( f2_grad_x1(x1,x2) ))
|
|
|
|
<span style="color: #008000">print</span>()
|
|
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The derivative of f2 w.r.t x2: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>( f2_grad_x2(x1,x2) ))
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The analytical derivative of f2 w.r.t x2: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>( f2_grad_x2(x1,x2) ))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable.</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="more-complicated-functions-using-the-elements-of-their-arguments-directly">More complicated functions using the elements of their arguments directly </h2>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f3</span>(x): <span style="color: #408080; font-style: italic"># Assumes x is an array of length 5 or higher</span>
|
|
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">2*</span>x[<span style="color: #666666">0</span>] <span style="color: #666666">+</span> <span style="color: #666666">3*</span>x[<span style="color: #666666">1</span>] <span style="color: #666666">+</span> <span style="color: #666666">5*</span>x[<span style="color: #666666">2</span>] <span style="color: #666666">+</span> <span style="color: #666666">7*</span>x[<span style="color: #666666">3</span>] <span style="color: #666666">+</span> <span style="color: #666666">11*</span>x[<span style="color: #666666">4</span>]<span style="color: #666666">**2</span>
|
|
|
|
f3_grad <span style="color: #666666">=</span> grad(f3)
|
|
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">0</span>,<span style="color: #666666">4</span>,<span style="color: #666666">5</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># Print the computed gradient:</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The computed gradient of f3 is: "</span>, f3_grad(x))
|
|
|
|
<span style="color: #408080; font-style: italic"># The analytical gradient is: (2, 3, 5, 7, 22*x[4])</span>
|
|
f3_grad_analytical <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">2</span>, <span style="color: #666666">3</span>, <span style="color: #666666">5</span>, <span style="color: #666666">7</span>, <span style="color: #666666">22*</span>x[<span style="color: #666666">4</span>]])
|
|
|
|
<span style="color: #408080; font-style: italic"># Print the analytical gradient:</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The analytical gradient of f3 is: "</span>, f3_grad_analytical)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>Note that in this case, when sending an array as input argument, the
|
|
output from Autograd is another array. This is the true gradient of
|
|
the function, as opposed to the function in the previous example. By
|
|
using arrays to represent the variables, the output from Autograd
|
|
might be easier to work with, as the output is closer to what one
|
|
could expect form a gradient-evaluting function.
|
|
</p>
|
|
|
|
<!-- !split -->
|
|
<h2 id="functions-using-mathematical-functions-from-numpy">Functions using mathematical functions from Numpy </h2>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f4</span>(x):
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sqrt(<span style="color: #666666">1+</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> np<span style="color: #666666">.</span>exp(x) <span style="color: #666666">+</span> np<span style="color: #666666">.</span>sin(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi<span style="color: #666666">*</span>x)
|
|
|
|
f4_grad <span style="color: #666666">=</span> grad(f4)
|
|
|
|
x <span style="color: #666666">=</span> <span style="color: #666666">2.7</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># Print the computed derivative:</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The computed derivative of f4 at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f4_grad(x)))
|
|
|
|
<span style="color: #408080; font-style: italic"># The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi</span>
|
|
f4_grad_analytical <span style="color: #666666">=</span> x<span style="color: #666666">/</span>np<span style="color: #666666">.</span>sqrt(<span style="color: #666666">1</span> <span style="color: #666666">+</span> x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> np<span style="color: #666666">.</span>exp(x) <span style="color: #666666">+</span> np<span style="color: #666666">.</span>cos(<span style="color: #666666">2*</span>np<span style="color: #666666">.</span>pi<span style="color: #666666">*</span>x)<span style="color: #666666">*2*</span>np<span style="color: #666666">.</span>pi
|
|
|
|
<span style="color: #408080; font-style: italic"># Print the analytical gradient:</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The analytical gradient of f4 at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f4_grad_analytical))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="more-autograd">More autograd </h2>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f5</span>(x):
|
|
<span style="color: #008000; font-weight: bold">if</span> x <span style="color: #666666">>=</span> <span style="color: #666666">0</span>:
|
|
<span style="color: #008000; font-weight: bold">return</span> x<span style="color: #666666">**2</span>
|
|
<span style="color: #008000; font-weight: bold">else</span>:
|
|
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">-3*</span>x <span style="color: #666666">+</span> <span style="color: #666666">1</span>
|
|
|
|
f5_grad <span style="color: #666666">=</span> grad(f5)
|
|
|
|
x <span style="color: #666666">=</span> <span style="color: #666666">2.7</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># Print the computed derivative:</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The computed derivative of f5 at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f5_grad(x)))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="and-with-loops">And with loops </h2>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f6_for</span>(x):
|
|
val <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">10</span>):
|
|
val <span style="color: #666666">=</span> val <span style="color: #666666">+</span> x<span style="color: #666666">**</span>i
|
|
<span style="color: #008000; font-weight: bold">return</span> val
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f6_while</span>(x):
|
|
val <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
|
i <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
|
<span style="color: #008000; font-weight: bold">while</span> i <span style="color: #666666"><</span> <span style="color: #666666">10</span>:
|
|
val <span style="color: #666666">=</span> val <span style="color: #666666">+</span> x<span style="color: #666666">**</span>i
|
|
i <span style="color: #666666">=</span> i <span style="color: #666666">+</span> <span style="color: #666666">1</span>
|
|
<span style="color: #008000; font-weight: bold">return</span> val
|
|
|
|
f6_for_grad <span style="color: #666666">=</span> grad(f6_for)
|
|
f6_while_grad <span style="color: #666666">=</span> grad(f6_while)
|
|
|
|
x <span style="color: #666666">=</span> <span style="color: #666666">0.5</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># Print the computed derivaties of f6_for and f6_while</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The computed derivative of f6_for at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f6_for_grad(x)))
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The computed derivative of f6_while at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f6_while_grad(x)))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
<span style="color: #408080; font-style: italic"># Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9</span>
|
|
<span style="color: #408080; font-style: italic"># The analytical derivative is: sum(i*x**(i-1)) </span>
|
|
f6_grad_analytical <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">10</span>):
|
|
f6_grad_analytical <span style="color: #666666">+=</span> i<span style="color: #666666">*</span>x<span style="color: #666666">**</span>(i<span style="color: #666666">-1</span>)
|
|
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The analytical derivative of f6 at x = </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(x,f6_grad_analytical))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="using-recursion">Using recursion </h2>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">f7</span>(n): <span style="color: #408080; font-style: italic"># Assume that n is an integer</span>
|
|
<span style="color: #008000; font-weight: bold">if</span> n <span style="color: #666666">==</span> <span style="color: #666666">1</span> <span style="color: #AA22FF; font-weight: bold">or</span> n <span style="color: #666666">==</span> <span style="color: #666666">0</span>:
|
|
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">1</span>
|
|
<span style="color: #008000; font-weight: bold">else</span>:
|
|
<span style="color: #008000; font-weight: bold">return</span> n<span style="color: #666666">*</span>f7(n<span style="color: #666666">-1</span>)
|
|
|
|
f7_grad <span style="color: #666666">=</span> grad(f7)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">2.0</span>
|
|
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The computed derivative of f7 at n = </span><span style="color: #BB6688; font-weight: bold">%d</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(n,f7_grad(n)))
|
|
|
|
<span style="color: #408080; font-style: italic"># The function f7 is an implementation of the factorial of n.</span>
|
|
<span style="color: #408080; font-style: italic"># By using the product rule, one can find that the derivative is:</span>
|
|
|
|
f7_grad_analytical <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #008000">int</span>(n)<span style="color: #666666">-1</span>):
|
|
tmp <span style="color: #666666">=</span> <span style="color: #666666">1</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> k <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #008000">int</span>(n)<span style="color: #666666">-1</span>):
|
|
<span style="color: #008000; font-weight: bold">if</span> k <span style="color: #666666">!=</span> i:
|
|
tmp <span style="color: #666666">*=</span> (n <span style="color: #666666">-</span> k)
|
|
f7_grad_analytical <span style="color: #666666">+=</span> tmp
|
|
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"The analytical derivative of f7 at n = </span><span style="color: #BB6688; font-weight: bold">%d</span><span style="color: #BA2121"> is: </span><span style="color: #BB6688; font-weight: bold">%g</span><span style="color: #BA2121">"</span><span style="color: #666666">%</span>(n,f7_grad_analytical))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input.</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="using-autograd-with-ols">Using Autograd with OLS </h2>
|
|
|
|
<p>We conclude the part on optmization by showing how we can make codes
|
|
for linear regression and logistic regression using <b>autograd</b>. The
|
|
first example shows results with ordinary leats squares.
|
|
</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #408080; font-style: italic"># Using Autograd to calculate gradients for OLS</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">CostOLS</span>(beta):
|
|
<span style="color: #008000; font-weight: bold">return</span> (<span style="color: #666666">1.0/</span>n)<span style="color: #666666">*</span>np<span style="color: #666666">.</span>sum((y<span style="color: #666666">-</span>X <span style="color: #666666">@</span> beta)<span style="color: #666666">**2</span>)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
|
x <span style="color: #666666">=</span> <span style="color: #666666">2*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n,<span style="color: #666666">1</span>)
|
|
y <span style="color: #666666">=</span> <span style="color: #666666">4+3*</span>x<span style="color: #666666">+</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(n,<span style="color: #666666">1</span>)
|
|
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((n,<span style="color: #666666">1</span>)), x]
|
|
XT_X <span style="color: #666666">=</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X
|
|
theta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(XT_X) <span style="color: #666666">@</span> (X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Own inversion"</span>)
|
|
<span style="color: #008000">print</span>(theta_linreg)
|
|
<span style="color: #408080; font-style: italic"># Hessian matrix</span>
|
|
H <span style="color: #666666">=</span> (<span style="color: #666666">2.0/</span>n)<span style="color: #666666">*</span> XT_X
|
|
EigValues, EigVectors <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>eig(H)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Eigenvalues of Hessian Matrix:</span><span style="color: #BB6688; font-weight: bold">{</span>EigValues<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)
|
|
eta <span style="color: #666666">=</span> <span style="color: #666666">1.0/</span>np<span style="color: #666666">.</span>max(EigValues)
|
|
Niterations <span style="color: #666666">=</span> <span style="color: #666666">1000</span>
|
|
<span style="color: #408080; font-style: italic"># define the gradient</span>
|
|
training_gradient <span style="color: #666666">=</span> grad(CostOLS)
|
|
|
|
<span style="color: #008000; font-weight: bold">for</span> <span style="color: #008000">iter</span> <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Niterations):
|
|
gradients <span style="color: #666666">=</span> training_gradient(theta)
|
|
theta <span style="color: #666666">-=</span> eta<span style="color: #666666">*</span>gradients
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own gd"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
|
|
xnew <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([[<span style="color: #666666">0</span>],[<span style="color: #666666">2</span>]])
|
|
Xnew <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)), xnew]
|
|
ypredict <span style="color: #666666">=</span> Xnew<span style="color: #666666">.</span>dot(theta)
|
|
ypredict2 <span style="color: #666666">=</span> Xnew<span style="color: #666666">.</span>dot(theta_linreg)
|
|
|
|
plt<span style="color: #666666">.</span>plot(xnew, ypredict, <span style="color: #BA2121">"r-"</span>)
|
|
plt<span style="color: #666666">.</span>plot(xnew, ypredict2, <span style="color: #BA2121">"b-"</span>)
|
|
plt<span style="color: #666666">.</span>plot(x, y ,<span style="color: #BA2121">'ro'</span>)
|
|
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">0</span>,<span style="color: #666666">2.0</span>,<span style="color: #666666">0</span>, <span style="color: #666666">15.0</span>])
|
|
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r'$x$'</span>)
|
|
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r'$y$'</span>)
|
|
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r'Random numbers '</span>)
|
|
plt<span style="color: #666666">.</span>show()
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="same-code-but-now-with-momentum-gradient-descent">Same code but now with momentum gradient descent </h2>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #408080; font-style: italic"># Using Autograd to calculate gradients for OLS</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">CostOLS</span>(beta):
|
|
<span style="color: #008000; font-weight: bold">return</span> (<span style="color: #666666">1.0/</span>n)<span style="color: #666666">*</span>np<span style="color: #666666">.</span>sum((y<span style="color: #666666">-</span>X <span style="color: #666666">@</span> beta)<span style="color: #666666">**2</span>)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
|
x <span style="color: #666666">=</span> <span style="color: #666666">2*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n,<span style="color: #666666">1</span>)
|
|
y <span style="color: #666666">=</span> <span style="color: #666666">4+3*</span>x<span style="color: #408080; font-style: italic">#+np.random.randn(n,1)</span>
|
|
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((n,<span style="color: #666666">1</span>)), x]
|
|
XT_X <span style="color: #666666">=</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X
|
|
theta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(XT_X) <span style="color: #666666">@</span> (X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Own inversion"</span>)
|
|
<span style="color: #008000">print</span>(theta_linreg)
|
|
<span style="color: #408080; font-style: italic"># Hessian matrix</span>
|
|
H <span style="color: #666666">=</span> (<span style="color: #666666">2.0/</span>n)<span style="color: #666666">*</span> XT_X
|
|
EigValues, EigVectors <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>eig(H)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Eigenvalues of Hessian Matrix:</span><span style="color: #BB6688; font-weight: bold">{</span>EigValues<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)
|
|
eta <span style="color: #666666">=</span> <span style="color: #666666">1.0/</span>np<span style="color: #666666">.</span>max(EigValues)
|
|
Niterations <span style="color: #666666">=</span> <span style="color: #666666">30</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># define the gradient</span>
|
|
training_gradient <span style="color: #666666">=</span> grad(CostOLS)
|
|
|
|
<span style="color: #008000; font-weight: bold">for</span> <span style="color: #008000">iter</span> <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Niterations):
|
|
gradients <span style="color: #666666">=</span> training_gradient(theta)
|
|
theta <span style="color: #666666">-=</span> eta<span style="color: #666666">*</span>gradients
|
|
<span style="color: #008000">print</span>(<span style="color: #008000">iter</span>,gradients[<span style="color: #666666">0</span>],gradients[<span style="color: #666666">1</span>])
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own gd"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
|
|
<span style="color: #408080; font-style: italic"># Now improve with momentum gradient descent</span>
|
|
change <span style="color: #666666">=</span> <span style="color: #666666">0.0</span>
|
|
delta_momentum <span style="color: #666666">=</span> <span style="color: #666666">0.3</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> <span style="color: #008000">iter</span> <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Niterations):
|
|
<span style="color: #408080; font-style: italic"># calculate gradient</span>
|
|
gradients <span style="color: #666666">=</span> training_gradient(theta)
|
|
<span style="color: #408080; font-style: italic"># calculate update</span>
|
|
new_change <span style="color: #666666">=</span> eta<span style="color: #666666">*</span>gradients<span style="color: #666666">+</span>delta_momentum<span style="color: #666666">*</span>change
|
|
<span style="color: #408080; font-style: italic"># take a step</span>
|
|
theta <span style="color: #666666">-=</span> new_change
|
|
<span style="color: #408080; font-style: italic"># save the change</span>
|
|
change <span style="color: #666666">=</span> new_change
|
|
<span style="color: #008000">print</span>(<span style="color: #008000">iter</span>,gradients[<span style="color: #666666">0</span>],gradients[<span style="color: #666666">1</span>])
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own gd wth momentum"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="including-stochastic-gradient-descent-with-autograd">Including Stochastic Gradient Descent with Autograd </h2>
|
|
<p>In this code we include the stochastic gradient descent approach discussed above. Note here that we specify which argument we are taking the derivative with respect to when using <b>autograd</b>.</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #408080; font-style: italic"># Using Autograd to calculate gradients using SGD</span>
|
|
<span style="color: #408080; font-style: italic"># OLS example</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #408080; font-style: italic"># Note change from previous example</span>
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">CostOLS</span>(y,X,theta):
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y<span style="color: #666666">-</span>X <span style="color: #666666">@</span> theta)<span style="color: #666666">**2</span>)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
|
x <span style="color: #666666">=</span> <span style="color: #666666">2*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n,<span style="color: #666666">1</span>)
|
|
y <span style="color: #666666">=</span> <span style="color: #666666">4+3*</span>x<span style="color: #666666">+</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(n,<span style="color: #666666">1</span>)
|
|
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((n,<span style="color: #666666">1</span>)), x]
|
|
XT_X <span style="color: #666666">=</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X
|
|
theta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(XT_X) <span style="color: #666666">@</span> (X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Own inversion"</span>)
|
|
<span style="color: #008000">print</span>(theta_linreg)
|
|
<span style="color: #408080; font-style: italic"># Hessian matrix</span>
|
|
H <span style="color: #666666">=</span> (<span style="color: #666666">2.0/</span>n)<span style="color: #666666">*</span> XT_X
|
|
EigValues, EigVectors <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>eig(H)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Eigenvalues of Hessian Matrix:</span><span style="color: #BB6688; font-weight: bold">{</span>EigValues<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)
|
|
eta <span style="color: #666666">=</span> <span style="color: #666666">1.0/</span>np<span style="color: #666666">.</span>max(EigValues)
|
|
Niterations <span style="color: #666666">=</span> <span style="color: #666666">1000</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># Note that we request the derivative wrt third argument (theta, 2 here)</span>
|
|
training_gradient <span style="color: #666666">=</span> grad(CostOLS,<span style="color: #666666">2</span>)
|
|
|
|
<span style="color: #008000; font-weight: bold">for</span> <span style="color: #008000">iter</span> <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Niterations):
|
|
gradients <span style="color: #666666">=</span> (<span style="color: #666666">1.0/</span>n)<span style="color: #666666">*</span>training_gradient(y, X, theta)
|
|
theta <span style="color: #666666">-=</span> eta<span style="color: #666666">*</span>gradients
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own gd"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
|
|
xnew <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([[<span style="color: #666666">0</span>],[<span style="color: #666666">2</span>]])
|
|
Xnew <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)), xnew]
|
|
ypredict <span style="color: #666666">=</span> Xnew<span style="color: #666666">.</span>dot(theta)
|
|
ypredict2 <span style="color: #666666">=</span> Xnew<span style="color: #666666">.</span>dot(theta_linreg)
|
|
|
|
plt<span style="color: #666666">.</span>plot(xnew, ypredict, <span style="color: #BA2121">"r-"</span>)
|
|
plt<span style="color: #666666">.</span>plot(xnew, ypredict2, <span style="color: #BA2121">"b-"</span>)
|
|
plt<span style="color: #666666">.</span>plot(x, y ,<span style="color: #BA2121">'ro'</span>)
|
|
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">0</span>,<span style="color: #666666">2.0</span>,<span style="color: #666666">0</span>, <span style="color: #666666">15.0</span>])
|
|
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r'$x$'</span>)
|
|
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r'$y$'</span>)
|
|
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r'Random numbers '</span>)
|
|
plt<span style="color: #666666">.</span>show()
|
|
|
|
n_epochs <span style="color: #666666">=</span> <span style="color: #666666">50</span>
|
|
M <span style="color: #666666">=</span> <span style="color: #666666">5</span> <span style="color: #408080; font-style: italic">#size of each minibatch</span>
|
|
m <span style="color: #666666">=</span> <span style="color: #008000">int</span>(n<span style="color: #666666">/</span>M) <span style="color: #408080; font-style: italic">#number of minibatches</span>
|
|
t0, t1 <span style="color: #666666">=</span> <span style="color: #666666">5</span>, <span style="color: #666666">50</span>
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">learning_schedule</span>(t):
|
|
<span style="color: #008000; font-weight: bold">return</span> t0<span style="color: #666666">/</span>(t<span style="color: #666666">+</span>t1)
|
|
|
|
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)
|
|
|
|
<span style="color: #008000; font-weight: bold">for</span> epoch <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n_epochs):
|
|
<span style="color: #408080; font-style: italic"># Can you figure out a better way of setting up the contributions to each batch?</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(m):
|
|
random_index <span style="color: #666666">=</span> M<span style="color: #666666">*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randint(m)
|
|
xi <span style="color: #666666">=</span> X[random_index:random_index<span style="color: #666666">+</span>M]
|
|
yi <span style="color: #666666">=</span> y[random_index:random_index<span style="color: #666666">+</span>M]
|
|
gradients <span style="color: #666666">=</span> (<span style="color: #666666">1.0/</span>M)<span style="color: #666666">*</span>training_gradient(yi, xi, theta)
|
|
eta <span style="color: #666666">=</span> learning_schedule(epoch<span style="color: #666666">*</span>m<span style="color: #666666">+</span>i)
|
|
theta <span style="color: #666666">=</span> theta <span style="color: #666666">-</span> eta<span style="color: #666666">*</span>gradients
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own sdg"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="same-code-but-now-with-momentum-gradient-descent">Same code but now with momentum gradient descent </h2>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #408080; font-style: italic"># Using Autograd to calculate gradients using SGD</span>
|
|
<span style="color: #408080; font-style: italic"># OLS example</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #408080; font-style: italic"># Note change from previous example</span>
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">CostOLS</span>(y,X,theta):
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y<span style="color: #666666">-</span>X <span style="color: #666666">@</span> theta)<span style="color: #666666">**2</span>)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
|
x <span style="color: #666666">=</span> <span style="color: #666666">2*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n,<span style="color: #666666">1</span>)
|
|
y <span style="color: #666666">=</span> <span style="color: #666666">4+3*</span>x<span style="color: #666666">+</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(n,<span style="color: #666666">1</span>)
|
|
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((n,<span style="color: #666666">1</span>)), x]
|
|
XT_X <span style="color: #666666">=</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X
|
|
theta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(XT_X) <span style="color: #666666">@</span> (X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Own inversion"</span>)
|
|
<span style="color: #008000">print</span>(theta_linreg)
|
|
<span style="color: #408080; font-style: italic"># Hessian matrix</span>
|
|
H <span style="color: #666666">=</span> (<span style="color: #666666">2.0/</span>n)<span style="color: #666666">*</span> XT_X
|
|
EigValues, EigVectors <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>eig(H)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Eigenvalues of Hessian Matrix:</span><span style="color: #BB6688; font-weight: bold">{</span>EigValues<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)
|
|
eta <span style="color: #666666">=</span> <span style="color: #666666">1.0/</span>np<span style="color: #666666">.</span>max(EigValues)
|
|
Niterations <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># Note that we request the derivative wrt third argument (theta, 2 here)</span>
|
|
training_gradient <span style="color: #666666">=</span> grad(CostOLS,<span style="color: #666666">2</span>)
|
|
|
|
<span style="color: #008000; font-weight: bold">for</span> <span style="color: #008000">iter</span> <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Niterations):
|
|
gradients <span style="color: #666666">=</span> (<span style="color: #666666">1.0/</span>n)<span style="color: #666666">*</span>training_gradient(y, X, theta)
|
|
theta <span style="color: #666666">-=</span> eta<span style="color: #666666">*</span>gradients
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own gd"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
|
|
|
|
n_epochs <span style="color: #666666">=</span> <span style="color: #666666">50</span>
|
|
M <span style="color: #666666">=</span> <span style="color: #666666">5</span> <span style="color: #408080; font-style: italic">#size of each minibatch</span>
|
|
m <span style="color: #666666">=</span> <span style="color: #008000">int</span>(n<span style="color: #666666">/</span>M) <span style="color: #408080; font-style: italic">#number of minibatches</span>
|
|
t0, t1 <span style="color: #666666">=</span> <span style="color: #666666">5</span>, <span style="color: #666666">50</span>
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">learning_schedule</span>(t):
|
|
<span style="color: #008000; font-weight: bold">return</span> t0<span style="color: #666666">/</span>(t<span style="color: #666666">+</span>t1)
|
|
|
|
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">2</span>,<span style="color: #666666">1</span>)
|
|
|
|
change <span style="color: #666666">=</span> <span style="color: #666666">0.0</span>
|
|
delta_momentum <span style="color: #666666">=</span> <span style="color: #666666">0.3</span>
|
|
|
|
<span style="color: #008000; font-weight: bold">for</span> epoch <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n_epochs):
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(m):
|
|
random_index <span style="color: #666666">=</span> M<span style="color: #666666">*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randint(m)
|
|
xi <span style="color: #666666">=</span> X[random_index:random_index<span style="color: #666666">+</span>M]
|
|
yi <span style="color: #666666">=</span> y[random_index:random_index<span style="color: #666666">+</span>M]
|
|
gradients <span style="color: #666666">=</span> (<span style="color: #666666">1.0/</span>M)<span style="color: #666666">*</span>training_gradient(yi, xi, theta)
|
|
eta <span style="color: #666666">=</span> learning_schedule(epoch<span style="color: #666666">*</span>m<span style="color: #666666">+</span>i)
|
|
<span style="color: #408080; font-style: italic"># calculate update</span>
|
|
new_change <span style="color: #666666">=</span> eta<span style="color: #666666">*</span>gradients<span style="color: #666666">+</span>delta_momentum<span style="color: #666666">*</span>change
|
|
<span style="color: #408080; font-style: italic"># take a step</span>
|
|
theta <span style="color: #666666">-=</span> new_change
|
|
<span style="color: #408080; font-style: italic"># save the change</span>
|
|
change <span style="color: #666666">=</span> new_change
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own sdg with momentum"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="similar-second-order-function-now-problem-but-now-with-adagrad">Similar (second order function now) problem but now with AdaGrad </h2>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #408080; font-style: italic"># Using Autograd to calculate gradients using AdaGrad and Stochastic Gradient descent</span>
|
|
<span style="color: #408080; font-style: italic"># OLS example</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #408080; font-style: italic"># Note change from previous example</span>
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">CostOLS</span>(y,X,theta):
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y<span style="color: #666666">-</span>X <span style="color: #666666">@</span> theta)<span style="color: #666666">**2</span>)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">1000</span>
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n,<span style="color: #666666">1</span>)
|
|
y <span style="color: #666666">=</span> <span style="color: #666666">2.0+3*</span>x <span style="color: #666666">+4*</span>x<span style="color: #666666">*</span>x
|
|
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((n,<span style="color: #666666">1</span>)), x, x<span style="color: #666666">*</span>x]
|
|
XT_X <span style="color: #666666">=</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X
|
|
theta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(XT_X) <span style="color: #666666">@</span> (X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Own inversion"</span>)
|
|
<span style="color: #008000">print</span>(theta_linreg)
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># Note that we request the derivative wrt third argument (theta, 2 here)</span>
|
|
training_gradient <span style="color: #666666">=</span> grad(CostOLS,<span style="color: #666666">2</span>)
|
|
<span style="color: #408080; font-style: italic"># Define parameters for Stochastic Gradient Descent</span>
|
|
n_epochs <span style="color: #666666">=</span> <span style="color: #666666">50</span>
|
|
M <span style="color: #666666">=</span> <span style="color: #666666">5</span> <span style="color: #408080; font-style: italic">#size of each minibatch</span>
|
|
m <span style="color: #666666">=</span> <span style="color: #008000">int</span>(n<span style="color: #666666">/</span>M) <span style="color: #408080; font-style: italic">#number of minibatches</span>
|
|
<span style="color: #408080; font-style: italic"># Guess for unknown parameters theta</span>
|
|
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">3</span>,<span style="color: #666666">1</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># Value for learning rate</span>
|
|
eta <span style="color: #666666">=</span> <span style="color: #666666">0.01</span>
|
|
<span style="color: #408080; font-style: italic"># Including AdaGrad parameter to avoid possible division by zero</span>
|
|
delta <span style="color: #666666">=</span> <span style="color: #666666">1e-8</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> epoch <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n_epochs):
|
|
Giter <span style="color: #666666">=</span> <span style="color: #666666">0.0</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(m):
|
|
random_index <span style="color: #666666">=</span> M<span style="color: #666666">*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randint(m)
|
|
xi <span style="color: #666666">=</span> X[random_index:random_index<span style="color: #666666">+</span>M]
|
|
yi <span style="color: #666666">=</span> y[random_index:random_index<span style="color: #666666">+</span>M]
|
|
gradients <span style="color: #666666">=</span> (<span style="color: #666666">1.0/</span>M)<span style="color: #666666">*</span>training_gradient(yi, xi, theta)
|
|
Giter <span style="color: #666666">+=</span> gradients<span style="color: #666666">*</span>gradients
|
|
update <span style="color: #666666">=</span> gradients<span style="color: #666666">*</span>eta<span style="color: #666666">/</span>(delta<span style="color: #666666">+</span>np<span style="color: #666666">.</span>sqrt(Giter))
|
|
theta <span style="color: #666666">-=</span> update
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own AdaGrad"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>Running this code we note an almost perfect agreement with the results from matrix inversion.</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="rmsprop-for-adaptive-learning-rate-with-stochastic-gradient-descent">RMSprop for adaptive learning rate with Stochastic Gradient Descent </h2>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #408080; font-style: italic"># Using Autograd to calculate gradients using RMSprop and Stochastic Gradient descent</span>
|
|
<span style="color: #408080; font-style: italic"># OLS example</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #408080; font-style: italic"># Note change from previous example</span>
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">CostOLS</span>(y,X,theta):
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y<span style="color: #666666">-</span>X <span style="color: #666666">@</span> theta)<span style="color: #666666">**2</span>)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">1000</span>
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n,<span style="color: #666666">1</span>)
|
|
y <span style="color: #666666">=</span> <span style="color: #666666">2.0+3*</span>x <span style="color: #666666">+4*</span>x<span style="color: #666666">*</span>x<span style="color: #408080; font-style: italic"># +np.random.randn(n,1)</span>
|
|
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((n,<span style="color: #666666">1</span>)), x, x<span style="color: #666666">*</span>x]
|
|
XT_X <span style="color: #666666">=</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X
|
|
theta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(XT_X) <span style="color: #666666">@</span> (X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Own inversion"</span>)
|
|
<span style="color: #008000">print</span>(theta_linreg)
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># Note that we request the derivative wrt third argument (theta, 2 here)</span>
|
|
training_gradient <span style="color: #666666">=</span> grad(CostOLS,<span style="color: #666666">2</span>)
|
|
<span style="color: #408080; font-style: italic"># Define parameters for Stochastic Gradient Descent</span>
|
|
n_epochs <span style="color: #666666">=</span> <span style="color: #666666">50</span>
|
|
M <span style="color: #666666">=</span> <span style="color: #666666">5</span> <span style="color: #408080; font-style: italic">#size of each minibatch</span>
|
|
m <span style="color: #666666">=</span> <span style="color: #008000">int</span>(n<span style="color: #666666">/</span>M) <span style="color: #408080; font-style: italic">#number of minibatches</span>
|
|
<span style="color: #408080; font-style: italic"># Guess for unknown parameters theta</span>
|
|
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">3</span>,<span style="color: #666666">1</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># Value for learning rate</span>
|
|
eta <span style="color: #666666">=</span> <span style="color: #666666">0.01</span>
|
|
<span style="color: #408080; font-style: italic"># Value for parameter rho</span>
|
|
rho <span style="color: #666666">=</span> <span style="color: #666666">0.99</span>
|
|
<span style="color: #408080; font-style: italic"># Including AdaGrad parameter to avoid possible division by zero</span>
|
|
delta <span style="color: #666666">=</span> <span style="color: #666666">1e-8</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> epoch <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n_epochs):
|
|
Giter <span style="color: #666666">=</span> <span style="color: #666666">0.0</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(m):
|
|
random_index <span style="color: #666666">=</span> M<span style="color: #666666">*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randint(m)
|
|
xi <span style="color: #666666">=</span> X[random_index:random_index<span style="color: #666666">+</span>M]
|
|
yi <span style="color: #666666">=</span> y[random_index:random_index<span style="color: #666666">+</span>M]
|
|
gradients <span style="color: #666666">=</span> (<span style="color: #666666">1.0/</span>M)<span style="color: #666666">*</span>training_gradient(yi, xi, theta)
|
|
<span style="color: #408080; font-style: italic"># Accumulated gradient</span>
|
|
<span style="color: #408080; font-style: italic"># Scaling with rho the new and the previous results</span>
|
|
Giter <span style="color: #666666">=</span> (rho<span style="color: #666666">*</span>Giter<span style="color: #666666">+</span>(<span style="color: #666666">1-</span>rho)<span style="color: #666666">*</span>gradients<span style="color: #666666">*</span>gradients)
|
|
<span style="color: #408080; font-style: italic"># Taking the diagonal only and inverting</span>
|
|
update <span style="color: #666666">=</span> gradients<span style="color: #666666">*</span>eta<span style="color: #666666">/</span>(delta<span style="color: #666666">+</span>np<span style="color: #666666">.</span>sqrt(Giter))
|
|
<span style="color: #408080; font-style: italic"># Hadamard product</span>
|
|
theta <span style="color: #666666">-=</span> update
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own RMSprop"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="and-finally-adam-https-arxiv-org-pdf-1412-6980-pdf">And finally <a href="https://arxiv.org/pdf/1412.6980.pdf" target="_blank">ADAM</a> </h2>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #408080; font-style: italic"># Using Autograd to calculate gradients using RMSprop and Stochastic Gradient descent</span>
|
|
<span style="color: #408080; font-style: italic"># OLS example</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">random</span> <span style="color: #008000; font-weight: bold">import</span> random, seed
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #408080; font-style: italic"># Note change from previous example</span>
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">CostOLS</span>(y,X,theta):
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y<span style="color: #666666">-</span>X <span style="color: #666666">@</span> theta)<span style="color: #666666">**2</span>)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">1000</span>
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n,<span style="color: #666666">1</span>)
|
|
y <span style="color: #666666">=</span> <span style="color: #666666">2.0+3*</span>x <span style="color: #666666">+4*</span>x<span style="color: #666666">*</span>x<span style="color: #408080; font-style: italic"># +np.random.randn(n,1)</span>
|
|
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>c_[np<span style="color: #666666">.</span>ones((n,<span style="color: #666666">1</span>)), x, x<span style="color: #666666">*</span>x]
|
|
XT_X <span style="color: #666666">=</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X
|
|
theta_linreg <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(XT_X) <span style="color: #666666">@</span> (X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Own inversion"</span>)
|
|
<span style="color: #008000">print</span>(theta_linreg)
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># Note that we request the derivative wrt third argument (theta, 2 here)</span>
|
|
training_gradient <span style="color: #666666">=</span> grad(CostOLS,<span style="color: #666666">2</span>)
|
|
<span style="color: #408080; font-style: italic"># Define parameters for Stochastic Gradient Descent</span>
|
|
n_epochs <span style="color: #666666">=</span> <span style="color: #666666">50</span>
|
|
M <span style="color: #666666">=</span> <span style="color: #666666">5</span> <span style="color: #408080; font-style: italic">#size of each minibatch</span>
|
|
m <span style="color: #666666">=</span> <span style="color: #008000">int</span>(n<span style="color: #666666">/</span>M) <span style="color: #408080; font-style: italic">#number of minibatches</span>
|
|
<span style="color: #408080; font-style: italic"># Guess for unknown parameters theta</span>
|
|
theta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randn(<span style="color: #666666">3</span>,<span style="color: #666666">1</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># Value for learning rate</span>
|
|
eta <span style="color: #666666">=</span> <span style="color: #666666">0.01</span>
|
|
<span style="color: #408080; font-style: italic"># Value for parameters beta1 and beta2, see https://arxiv.org/abs/1412.6980</span>
|
|
beta1 <span style="color: #666666">=</span> <span style="color: #666666">0.9</span>
|
|
beta2 <span style="color: #666666">=</span> <span style="color: #666666">0.999</span>
|
|
<span style="color: #408080; font-style: italic"># Including AdaGrad parameter to avoid possible division by zero</span>
|
|
delta <span style="color: #666666">=</span> <span style="color: #666666">1e-7</span>
|
|
<span style="color: #008000">iter</span> <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> epoch <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n_epochs):
|
|
first_moment <span style="color: #666666">=</span> <span style="color: #666666">0.0</span>
|
|
second_moment <span style="color: #666666">=</span> <span style="color: #666666">0.0</span>
|
|
<span style="color: #008000">iter</span> <span style="color: #666666">+=</span> <span style="color: #666666">1</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(m):
|
|
random_index <span style="color: #666666">=</span> M<span style="color: #666666">*</span>np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>randint(m)
|
|
xi <span style="color: #666666">=</span> X[random_index:random_index<span style="color: #666666">+</span>M]
|
|
yi <span style="color: #666666">=</span> y[random_index:random_index<span style="color: #666666">+</span>M]
|
|
gradients <span style="color: #666666">=</span> (<span style="color: #666666">1.0/</span>M)<span style="color: #666666">*</span>training_gradient(yi, xi, theta)
|
|
<span style="color: #408080; font-style: italic"># Computing moments first</span>
|
|
first_moment <span style="color: #666666">=</span> beta1<span style="color: #666666">*</span>first_moment <span style="color: #666666">+</span> (<span style="color: #666666">1-</span>beta1)<span style="color: #666666">*</span>gradients
|
|
second_moment <span style="color: #666666">=</span> beta2<span style="color: #666666">*</span>second_moment<span style="color: #666666">+</span>(<span style="color: #666666">1-</span>beta2)<span style="color: #666666">*</span>gradients<span style="color: #666666">*</span>gradients
|
|
first_term <span style="color: #666666">=</span> first_moment<span style="color: #666666">/</span>(<span style="color: #666666">1.0-</span>beta1<span style="color: #666666">**</span><span style="color: #008000">iter</span>)
|
|
second_term <span style="color: #666666">=</span> second_moment<span style="color: #666666">/</span>(<span style="color: #666666">1.0-</span>beta2<span style="color: #666666">**</span><span style="color: #008000">iter</span>)
|
|
<span style="color: #408080; font-style: italic"># Scaling with rho the new and the previous results</span>
|
|
update <span style="color: #666666">=</span> eta<span style="color: #666666">*</span>first_term<span style="color: #666666">/</span>(np<span style="color: #666666">.</span>sqrt(second_term)<span style="color: #666666">+</span>delta)
|
|
theta <span style="color: #666666">-=</span> update
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"theta from own ADAM"</span>)
|
|
<span style="color: #008000">print</span>(theta)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="and-logistic-regression">And Logistic Regression </h2>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">autograd.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">autograd</span> <span style="color: #008000; font-weight: bold">import</span> grad
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">sigmoid</span>(x):
|
|
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">0.5</span> <span style="color: #666666">*</span> (np<span style="color: #666666">.</span>tanh(x <span style="color: #666666">/</span> <span style="color: #666666">2.</span>) <span style="color: #666666">+</span> <span style="color: #666666">1</span>)
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">logistic_predictions</span>(weights, inputs):
|
|
<span style="color: #408080; font-style: italic"># Outputs probability of a label being true according to logistic model.</span>
|
|
<span style="color: #008000; font-weight: bold">return</span> sigmoid(np<span style="color: #666666">.</span>dot(inputs, weights))
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">training_loss</span>(weights):
|
|
<span style="color: #408080; font-style: italic"># Training loss is the negative log-likelihood of the training labels.</span>
|
|
preds <span style="color: #666666">=</span> logistic_predictions(weights, inputs)
|
|
label_probabilities <span style="color: #666666">=</span> preds <span style="color: #666666">*</span> targets <span style="color: #666666">+</span> (<span style="color: #666666">1</span> <span style="color: #666666">-</span> preds) <span style="color: #666666">*</span> (<span style="color: #666666">1</span> <span style="color: #666666">-</span> targets)
|
|
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">-</span>np<span style="color: #666666">.</span>sum(np<span style="color: #666666">.</span>log(label_probabilities))
|
|
|
|
<span style="color: #408080; font-style: italic"># Build a toy dataset.</span>
|
|
inputs <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([[<span style="color: #666666">0.52</span>, <span style="color: #666666">1.12</span>, <span style="color: #666666">0.77</span>],
|
|
[<span style="color: #666666">0.88</span>, <span style="color: #666666">-1.08</span>, <span style="color: #666666">0.15</span>],
|
|
[<span style="color: #666666">0.52</span>, <span style="color: #666666">0.06</span>, <span style="color: #666666">-1.30</span>],
|
|
[<span style="color: #666666">0.74</span>, <span style="color: #666666">-2.49</span>, <span style="color: #666666">1.39</span>]])
|
|
targets <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #008000; font-weight: bold">True</span>, <span style="color: #008000; font-weight: bold">True</span>, <span style="color: #008000; font-weight: bold">False</span>, <span style="color: #008000; font-weight: bold">True</span>])
|
|
|
|
<span style="color: #408080; font-style: italic"># Define a function that returns gradients of training loss using Autograd.</span>
|
|
training_gradient_fun <span style="color: #666666">=</span> grad(training_loss)
|
|
|
|
<span style="color: #408080; font-style: italic"># Optimize weights using gradient descent.</span>
|
|
weights <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">0.0</span>, <span style="color: #666666">0.0</span>, <span style="color: #666666">0.0</span>])
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Initial loss:"</span>, training_loss(weights))
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">100</span>):
|
|
weights <span style="color: #666666">-=</span> training_gradient_fun(weights) <span style="color: #666666">*</span> <span style="color: #666666">0.01</span>
|
|
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Trained loss:"</span>, training_loss(weights))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<h2 id="introducing-jax-https-jax-readthedocs-io-en-latest">Introducing <a href="https://jax.readthedocs.io/en/latest/" target="_blank">JAX</a> </h2>
|
|
|
|
<p>Presently, instead of using <b>autograd</b>, we recommend using <a href="https://jax.readthedocs.io/en/latest/" target="_blank">JAX</a></p>
|
|
|
|
<p><b>JAX</b> is Autograd and <a href="https://www.tensorflow.org/xla" target="_blank">XLA (Accelerated Linear Algebra))</a>,
|
|
brought together for high-performance numerical computing and machine learning research.
|
|
It provides composable transformations of Python+NumPy programs: differentiate, vectorize, parallelize, Just-In-Time compile to GPU/TPU, and more.
|
|
</p>
|
|
<h3 id="getting-started-with-jax-note-the-way-we-import-numpy">Getting started with Jax, note the way we import numpy </h3>
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">jax</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">jax.numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">jnp</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">jax</span> <span style="color: #008000; font-weight: bold">import</span> grad <span style="color: #008000; font-weight: bold">as</span> jax_grad
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<h3 id="a-warm-up-example">A warm-up example </h3>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">function</span>(x):
|
|
<span style="color: #008000; font-weight: bold">return</span> x<span style="color: #666666">**2</span>
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">analytical_gradient</span>(x):
|
|
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">2*</span>x
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">gradient_descent</span>(starting_point, learning_rate, num_iterations, solver<span style="color: #666666">=</span><span style="color: #BA2121">"analytical"</span>):
|
|
x <span style="color: #666666">=</span> starting_point
|
|
trajectory_x <span style="color: #666666">=</span> [x]
|
|
trajectory_y <span style="color: #666666">=</span> [function(x)]
|
|
|
|
<span style="color: #008000; font-weight: bold">if</span> solver <span style="color: #666666">==</span> <span style="color: #BA2121">"analytical"</span>:
|
|
grad <span style="color: #666666">=</span> analytical_gradient
|
|
<span style="color: #008000; font-weight: bold">elif</span> solver <span style="color: #666666">==</span> <span style="color: #BA2121">"jax"</span>:
|
|
grad <span style="color: #666666">=</span> jax_grad(function)
|
|
x <span style="color: #666666">=</span> jnp<span style="color: #666666">.</span>float64(x)
|
|
learning_rate <span style="color: #666666">=</span> jnp<span style="color: #666666">.</span>float64(learning_rate)
|
|
|
|
<span style="color: #008000; font-weight: bold">for</span> _ <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(num_iterations):
|
|
|
|
x <span style="color: #666666">=</span> x <span style="color: #666666">-</span> learning_rate <span style="color: #666666">*</span> grad(x)
|
|
trajectory_x<span style="color: #666666">.</span>append(x)
|
|
trajectory_y<span style="color: #666666">.</span>append(function(x))
|
|
|
|
<span style="color: #008000; font-weight: bold">return</span> trajectory_x, trajectory_y
|
|
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">-5</span>, <span style="color: #666666">5</span>, <span style="color: #666666">100</span>)
|
|
plt<span style="color: #666666">.</span>plot(x, function(x), label<span style="color: #666666">=</span><span style="color: #BA2121">"f(x)"</span>)
|
|
|
|
descent_x, descent_y <span style="color: #666666">=</span> gradient_descent(<span style="color: #666666">5</span>, <span style="color: #666666">0.1</span>, <span style="color: #666666">10</span>, solver<span style="color: #666666">=</span><span style="color: #BA2121">"analytical"</span>)
|
|
jax_descend_x, jax_descend_y <span style="color: #666666">=</span> gradient_descent(<span style="color: #666666">5</span>, <span style="color: #666666">0.1</span>, <span style="color: #666666">10</span>, solver<span style="color: #666666">=</span><span style="color: #BA2121">"jax"</span>)
|
|
|
|
plt<span style="color: #666666">.</span>plot(descent_x, descent_y, label<span style="color: #666666">=</span><span style="color: #BA2121">"Gradient descent"</span>, marker<span style="color: #666666">=</span><span style="color: #BA2121">"o"</span>)
|
|
plt<span style="color: #666666">.</span>plot(jax_descend_x, jax_descend_y, label<span style="color: #666666">=</span><span style="color: #BA2121">"JAX"</span>, marker<span style="color: #666666">=</span><span style="color: #BA2121">"x"</span>)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<h3 id="a-more-advanced-example">A more advanced example </h3>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;">backend <span style="color: #666666">=</span> np
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">function</span>(x):
|
|
<span style="color: #008000; font-weight: bold">return</span> x<span style="color: #666666">*</span>backend<span style="color: #666666">.</span>sin(x<span style="color: #666666">**2</span> <span style="color: #666666">+</span> <span style="color: #666666">1</span>)
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">analytical_gradient</span>(x):
|
|
<span style="color: #008000; font-weight: bold">return</span> backend<span style="color: #666666">.</span>sin(x<span style="color: #666666">**2</span> <span style="color: #666666">+</span> <span style="color: #666666">1</span>) <span style="color: #666666">+</span> <span style="color: #666666">2*</span>x<span style="color: #666666">**2*</span>backend<span style="color: #666666">.</span>cos(x<span style="color: #666666">**2</span> <span style="color: #666666">+</span> <span style="color: #666666">1</span>)
|
|
|
|
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">-5</span>, <span style="color: #666666">5</span>, <span style="color: #666666">100</span>)
|
|
plt<span style="color: #666666">.</span>plot(x, function(x), label<span style="color: #666666">=</span><span style="color: #BA2121">"f(x)"</span>)
|
|
|
|
descent_x, descent_y <span style="color: #666666">=</span> gradient_descent(<span style="color: #666666">1</span>, <span style="color: #666666">0.01</span>, <span style="color: #666666">300</span>, solver<span style="color: #666666">=</span><span style="color: #BA2121">"analytical"</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># Change the backend to JAX</span>
|
|
backend <span style="color: #666666">=</span> jnp
|
|
jax_descend_x, jax_descend_y <span style="color: #666666">=</span> gradient_descent(<span style="color: #666666">1</span>, <span style="color: #666666">0.01</span>, <span style="color: #666666">300</span>, solver<span style="color: #666666">=</span><span style="color: #BA2121">"jax"</span>)
|
|
|
|
plt<span style="color: #666666">.</span>scatter(descent_x, descent_y, label<span style="color: #666666">=</span><span style="color: #BA2121">"Gradient descent"</span>, marker<span style="color: #666666">=</span><span style="color: #BA2121">"v"</span>, s<span style="color: #666666">=10</span>, color<span style="color: #666666">=</span><span style="color: #BA2121">"red"</span>)
|
|
plt<span style="color: #666666">.</span>scatter(jax_descend_x, jax_descend_y, label<span style="color: #666666">=</span><span style="color: #BA2121">"JAX"</span>, marker<span style="color: #666666">=</span><span style="color: #BA2121">"x"</span>, s<span style="color: #666666">=5</span>, color<span style="color: #666666">=</span><span style="color: #BA2121">"black"</span>)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="introduction-to-neural-networks">Introduction to Neural networks </h2>
|
|
|
|
<p>Artificial neural networks are computational systems that can learn to
|
|
perform tasks by considering examples, generally without being
|
|
programmed with any task-specific rules. It is supposed to mimic a
|
|
biological system, wherein neurons interact by sending signals in the
|
|
form of mathematical functions between layers. All layers can contain
|
|
an arbitrary number of neurons, and each connection is represented by
|
|
a weight variable.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="artificial-neurons">Artificial neurons </h2>
|
|
|
|
<p>The field of artificial neural networks has a long history of
|
|
development, and is closely connected with the advancement of computer
|
|
science and computers in general. A model of artificial neurons was
|
|
first developed by McCulloch and Pitts in 1943 to study signal
|
|
processing in the brain and has later been refined by others. The
|
|
general idea is to mimic neural networks in the human brain, which is
|
|
composed of billions of neurons that communicate with each other by
|
|
sending electrical signals. Each neuron accumulates its incoming
|
|
signals, which must exceed an activation threshold to yield an
|
|
output. If the threshold is not overcome, the neuron remains inactive,
|
|
i.e. has zero output.
|
|
</p>
|
|
|
|
<p>This behaviour has inspired a simple mathematical model for an artificial neuron.</p>
|
|
|
|
$$
|
|
\begin{equation}
|
|
y = f\left(\sum_{i=1}^n w_ix_i\right) = f(u)
|
|
\label{artificialNeuron}
|
|
\end{equation}
|
|
$$
|
|
|
|
<p>Here, the output \( y \) of the neuron is the value of its activation function, which have as input
|
|
a weighted sum of signals \( x_i, \dots ,x_n \) received by \( n \) other neurons.
|
|
</p>
|
|
|
|
<p>Conceptually, it is helpful to divide neural networks into four
|
|
categories:
|
|
</p>
|
|
<ol>
|
|
<li> general purpose neural networks for supervised learning,</li>
|
|
<li> neural networks designed specifically for image processing, the most prominent example of this class being Convolutional Neural Networks (CNNs),</li>
|
|
<li> neural networks for sequential data such as Recurrent Neural Networks (RNNs), and</li>
|
|
<li> neural networks for unsupervised learning such as Deep Boltzmann Machines.</li>
|
|
</ol>
|
|
<p>In natural science, DNNs and CNNs have already found numerous
|
|
applications. In statistical physics, they have been applied to detect
|
|
phase transitions in 2D Ising and Potts models, lattice gauge
|
|
theories, and different phases of polymers, or solving the
|
|
Navier-Stokes equation in weather forecasting. Deep learning has also
|
|
found interesting applications in quantum physics. Various quantum
|
|
phase transitions can be detected and studied using DNNs and CNNs,
|
|
topological phases, and even non-equilibrium many-body
|
|
localization. Representing quantum states as DNNs quantum state
|
|
tomography are among some of the impressive achievements to reveal the
|
|
potential of DNNs to facilitate the study of quantum systems.
|
|
</p>
|
|
|
|
<p>In quantum information theory, it has been shown that one can perform
|
|
gate decompositions with the help of neural.
|
|
</p>
|
|
|
|
<p>The applications are not limited to the natural sciences. There is a
|
|
plethora of applications in essentially all disciplines, from the
|
|
humanities to life science and medicine.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="neural-network-types">Neural network types </h2>
|
|
|
|
<p>An artificial neural network (ANN), is a computational model that
|
|
consists of layers of connected neurons, or nodes or units. We will
|
|
refer to these interchangeably as units or nodes, and sometimes as
|
|
neurons.
|
|
</p>
|
|
|
|
<p>It is supposed to mimic a biological nervous system by letting each
|
|
neuron interact with other neurons by sending signals in the form of
|
|
mathematical functions between layers. A wide variety of different
|
|
ANNs have been developed, but most of them consist of an input layer,
|
|
an output layer and eventual layers in-between, called <em>hidden
|
|
layers</em>. All layers can contain an arbitrary number of nodes, and each
|
|
connection between two nodes is associated with a weight variable.
|
|
</p>
|
|
|
|
<p>Neural networks (also called neural nets) are neural-inspired
|
|
nonlinear models for supervised learning. As we will see, neural nets
|
|
can be viewed as natural, more powerful extensions of supervised
|
|
learning methods such as linear and logistic regression and soft-max
|
|
methods we discussed earlier.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="feed-forward-neural-networks">Feed-forward neural networks </h2>
|
|
|
|
<p>The feed-forward neural network (FFNN) was the first and simplest type
|
|
of ANNs that were devised. In this network, the information moves in
|
|
only one direction: forward through the layers.
|
|
</p>
|
|
|
|
<p>Nodes are represented by circles, while the arrows display the
|
|
connections between the nodes, including the direction of information
|
|
flow. Additionally, each arrow corresponds to a weight variable
|
|
(figure to come). We observe that each node in a layer is connected
|
|
to <em>all</em> nodes in the subsequent layer, making this a so-called
|
|
<em>fully-connected</em> FFNN.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="convolutional-neural-network">Convolutional Neural Network </h2>
|
|
|
|
<p>A different variant of FFNNs are <em>convolutional neural networks</em>
|
|
(CNNs), which have a connectivity pattern inspired by the animal
|
|
visual cortex. Individual neurons in the visual cortex only respond to
|
|
stimuli from small sub-regions of the visual field, called a receptive
|
|
field. This makes the neurons well-suited to exploit the strong
|
|
spatially local correlation present in natural images. The response of
|
|
each neuron can be approximated mathematically as a convolution
|
|
operation. (figure to come)
|
|
</p>
|
|
|
|
<p>Convolutional neural networks emulate the behaviour of neurons in the
|
|
visual cortex by enforcing a <em>local</em> connectivity pattern between
|
|
nodes of adjacent layers: Each node in a convolutional layer is
|
|
connected only to a subset of the nodes in the previous layer, in
|
|
contrast to the fully-connected FFNN. Often, CNNs consist of several
|
|
convolutional layers that learn local features of the input, with a
|
|
fully-connected layer at the end, which gathers all the local data and
|
|
produces the outputs. They have wide applications in image and video
|
|
recognition.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="recurrent-neural-networks">Recurrent neural networks </h2>
|
|
|
|
<p>So far we have only mentioned ANNs where information flows in one
|
|
direction: forward. <em>Recurrent neural networks</em> on the other hand,
|
|
have connections between nodes that form directed <em>cycles</em>. This
|
|
creates a form of internal memory which are able to capture
|
|
information on what has been calculated before; the output is
|
|
dependent on the previous computations. Recurrent NNs make use of
|
|
sequential information by performing the same task for every element
|
|
in a sequence, where each element depends on previous elements. An
|
|
example of such information is sentences, making recurrent NNs
|
|
especially well-suited for handwriting and speech recognition.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="other-types-of-networks">Other types of networks </h2>
|
|
|
|
<p>There are many other kinds of ANNs that have been developed. One type
|
|
that is specifically designed for interpolation in multidimensional
|
|
space is the radial basis function (RBF) network. RBFs are typically
|
|
made up of three layers: an input layer, a hidden layer with
|
|
non-linear radial symmetric activation functions and a linear output
|
|
layer (''linear'' here means that each node in the output layer has a
|
|
linear activation function). The layers are normally fully-connected
|
|
and there are no cycles, thus RBFs can be viewed as a type of
|
|
fully-connected FFNN. They are however usually treated as a separate
|
|
type of NN due the unusual activation functions.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="multilayer-perceptrons">Multilayer perceptrons </h2>
|
|
|
|
<p>One uses often so-called fully-connected feed-forward neural networks
|
|
with three or more layers (an input layer, one or more hidden layers
|
|
and an output layer) consisting of neurons that have non-linear
|
|
activation functions.
|
|
</p>
|
|
|
|
<p>Such networks are often called <em>multilayer perceptrons</em> (MLPs).</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="why-multilayer-perceptrons">Why multilayer perceptrons? </h2>
|
|
|
|
<p>According to the <em>Universal approximation theorem</em>, a feed-forward
|
|
neural network with just a single hidden layer containing a finite
|
|
number of neurons can approximate a continuous multidimensional
|
|
function to arbitrary accuracy, assuming the activation function for
|
|
the hidden layer is a <b>non-constant, bounded and
|
|
monotonically-increasing continuous function</b>.
|
|
</p>
|
|
|
|
<p>Note that the requirements on the activation function only applies to
|
|
the hidden layer, the output nodes are always assumed to be linear, so
|
|
as to not restrict the range of output values.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="illustration-of-a-single-perceptron-model-and-a-multi-perceptron-model">Illustration of a single perceptron model and a multi-perceptron model </h2>
|
|
|
|
<center> <!-- FIGURE -->
|
|
<hr class="figure">
|
|
<center>
|
|
<p class="caption">Figure 1: In a) we show a single perceptron model while in b) we dispay a network with two hidden layers, an input layer and an output layer. </p>
|
|
</center>
|
|
<p><img src="figures/nns.png" width="600" align="bottom"></p>
|
|
</center>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="examples-of-xor-or-and-and-gates">Examples of XOR, OR and AND gates </h2>
|
|
|
|
<p>Let us first try to fit various gates using standard linear
|
|
regression. The gates we are thinking of are the classical XOR, OR and
|
|
AND gates, well-known elements in computer science. The tables here
|
|
show how we can set up the inputs \( x_1 \) and \( x_2 \) in order to yield a
|
|
specific target \( y_i \).
|
|
</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #BA2121; font-style: italic">"""</span>
|
|
<span style="color: #BA2121; font-style: italic">Simple code that tests XOR, OR and AND gates with linear regression</span>
|
|
<span style="color: #BA2121; font-style: italic">"""</span>
|
|
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #408080; font-style: italic"># Design matrix</span>
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([ [<span style="color: #666666">1</span>, <span style="color: #666666">0</span>, <span style="color: #666666">0</span>], [<span style="color: #666666">1</span>, <span style="color: #666666">0</span>, <span style="color: #666666">1</span>], [<span style="color: #666666">1</span>, <span style="color: #666666">1</span>, <span style="color: #666666">0</span>],[<span style="color: #666666">1</span>, <span style="color: #666666">1</span>, <span style="color: #666666">1</span>]],dtype<span style="color: #666666">=</span>np<span style="color: #666666">.</span>float64)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The X.TX matrix:</span><span style="color: #BB6688; font-weight: bold">{</span>X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
Xinv <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The invers of X.TX matrix:</span><span style="color: #BB6688; font-weight: bold">{</span>Xinv<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># The XOR gate </span>
|
|
yXOR <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array( [ <span style="color: #666666">0</span>, <span style="color: #666666">1</span> ,<span style="color: #666666">1</span>, <span style="color: #666666">0</span>])
|
|
ThetaXOR <span style="color: #666666">=</span> Xinv <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> yXOR
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The values of theta for the XOR gate:</span><span style="color: #BB6688; font-weight: bold">{</span>ThetaXOR<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The linear regression prediction for the XOR gate:</span><span style="color: #BB6688; font-weight: bold">{</span>X <span style="color: #666666">@</span> ThetaXOR<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># The OR gate </span>
|
|
yOR <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array( [ <span style="color: #666666">0</span>, <span style="color: #666666">1</span> ,<span style="color: #666666">1</span>, <span style="color: #666666">1</span>])
|
|
ThetaOR <span style="color: #666666">=</span> Xinv <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> yOR
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The values of theta for the OR gate:</span><span style="color: #BB6688; font-weight: bold">{</span>ThetaOR<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The linear regression prediction for the OR gate:</span><span style="color: #BB6688; font-weight: bold">{</span>X <span style="color: #666666">@</span> ThetaOR<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># The OR gate </span>
|
|
yAND <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array( [ <span style="color: #666666">0</span>, <span style="color: #666666">0</span> ,<span style="color: #666666">0</span>, <span style="color: #666666">1</span>])
|
|
ThetaAND <span style="color: #666666">=</span> Xinv <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> yAND
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The values of theta for the AND gate:</span><span style="color: #BB6688; font-weight: bold">{</span>ThetaAND<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The linear regression prediction for the AND gate:</span><span style="color: #BB6688; font-weight: bold">{</span>X <span style="color: #666666">@</span> ThetaAND<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>What is happening here?</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="does-logistic-regression-do-a-better-job">Does Logistic Regression do a better Job? </h2>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #BA2121; font-style: italic">"""</span>
|
|
<span style="color: #BA2121; font-style: italic">Simple code that tests XOR and OR gates with linear regression</span>
|
|
<span style="color: #BA2121; font-style: italic">and logistic regression</span>
|
|
<span style="color: #BA2121; font-style: italic">"""</span>
|
|
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LogisticRegression
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
|
|
<span style="color: #408080; font-style: italic"># Design matrix</span>
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([ [<span style="color: #666666">1</span>, <span style="color: #666666">0</span>, <span style="color: #666666">0</span>], [<span style="color: #666666">1</span>, <span style="color: #666666">0</span>, <span style="color: #666666">1</span>], [<span style="color: #666666">1</span>, <span style="color: #666666">1</span>, <span style="color: #666666">0</span>],[<span style="color: #666666">1</span>, <span style="color: #666666">1</span>, <span style="color: #666666">1</span>]],dtype<span style="color: #666666">=</span>np<span style="color: #666666">.</span>float64)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The X.TX matrix:</span><span style="color: #BB6688; font-weight: bold">{</span>X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
Xinv <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The invers of X.TX matrix:</span><span style="color: #BB6688; font-weight: bold">{</span>Xinv<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># The XOR gate </span>
|
|
yXOR <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array( [ <span style="color: #666666">0</span>, <span style="color: #666666">1</span> ,<span style="color: #666666">1</span>, <span style="color: #666666">0</span>])
|
|
ThetaXOR <span style="color: #666666">=</span> Xinv <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> yXOR
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The values of theta for the XOR gate:</span><span style="color: #BB6688; font-weight: bold">{</span>ThetaXOR<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The linear regression prediction for the XOR gate:</span><span style="color: #BB6688; font-weight: bold">{</span>X <span style="color: #666666">@</span> ThetaXOR<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># The OR gate </span>
|
|
yOR <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array( [ <span style="color: #666666">0</span>, <span style="color: #666666">1</span> ,<span style="color: #666666">1</span>, <span style="color: #666666">1</span>])
|
|
ThetaOR <span style="color: #666666">=</span> Xinv <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> yOR
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The values of theta for the OR gate:</span><span style="color: #BB6688; font-weight: bold">{</span>ThetaOR<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The linear regression prediction for the OR gate:</span><span style="color: #BB6688; font-weight: bold">{</span>X <span style="color: #666666">@</span> ThetaOR<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># The OR gate </span>
|
|
yAND <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array( [ <span style="color: #666666">0</span>, <span style="color: #666666">0</span> ,<span style="color: #666666">0</span>, <span style="color: #666666">1</span>])
|
|
ThetaAND <span style="color: #666666">=</span> Xinv <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> yAND
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The values of theta for the AND gate:</span><span style="color: #BB6688; font-weight: bold">{</span>ThetaAND<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"The linear regression prediction for the AND gate:</span><span style="color: #BB6688; font-weight: bold">{</span>X <span style="color: #666666">@</span> ThetaAND<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># Now we change to logistic regression</span>
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># Logistic Regression</span>
|
|
logreg <span style="color: #666666">=</span> LogisticRegression()
|
|
logreg<span style="color: #666666">.</span>fit(X, yOR)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Test set accuracy with Logistic Regression for OR gate: </span><span style="color: #BB6688; font-weight: bold">{:.2f}</span><span style="color: #BA2121">"</span><span style="color: #666666">.</span>format(logreg<span style="color: #666666">.</span>score(X,yOR)))
|
|
|
|
logreg<span style="color: #666666">.</span>fit(X, yXOR)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Test set accuracy with Logistic Regression for XOR gate: </span><span style="color: #BB6688; font-weight: bold">{:.2f}</span><span style="color: #BA2121">"</span><span style="color: #666666">.</span>format(logreg<span style="color: #666666">.</span>score(X,yXOR)))
|
|
|
|
|
|
logreg<span style="color: #666666">.</span>fit(X, yAND)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Test set accuracy with Logistic Regression for AND gate: </span><span style="color: #BB6688; font-weight: bold">{:.2f}</span><span style="color: #BA2121">"</span><span style="color: #666666">.</span>format(logreg<span style="color: #666666">.</span>score(X,yAND)))
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>Not exactly impressive, but somewhat better.</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="adding-neural-networks">Adding Neural Networks </h2>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #408080; font-style: italic"># and now neural networks with Scikit-Learn and the XOR</span>
|
|
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.neural_network</span> <span style="color: #008000; font-weight: bold">import</span> MLPClassifier
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.datasets</span> <span style="color: #008000; font-weight: bold">import</span> make_classification
|
|
X, yXOR <span style="color: #666666">=</span> make_classification(n_samples<span style="color: #666666">=100</span>, random_state<span style="color: #666666">=1</span>)
|
|
FFNN <span style="color: #666666">=</span> MLPClassifier(random_state<span style="color: #666666">=1</span>, max_iter<span style="color: #666666">=300</span>)<span style="color: #666666">.</span>fit(X, yXOR)
|
|
FFNN<span style="color: #666666">.</span>predict_proba(X)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Test set accuracy with Feed Forward Neural Network for XOR gate:</span><span style="color: #BB6688; font-weight: bold">{</span>FFNN<span style="color: #666666">.</span>score(X, yXOR)<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="mathematical-model">Mathematical model </h2>
|
|
|
|
<p>The output \( y \) is produced via the activation function \( f \)</p>
|
|
$$
|
|
y = f\left(\sum_{i=1}^n w_ix_i + b_i\right) = f(z),
|
|
$$
|
|
|
|
<p>This function receives \( x_i \) as inputs.
|
|
Here the activation \( z=(\sum_{i=1}^n w_ix_i+b_i) \).
|
|
In an FFNN of such neurons, the <em>inputs</em> \( x_i \) are the <em>outputs</em> of
|
|
the neurons in the preceding layer. Furthermore, an MLP is
|
|
fully-connected, which means that each neuron receives a weighted sum
|
|
of the outputs of <em>all</em> neurons in the previous layer.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="mathematical-model">Mathematical model </h2>
|
|
|
|
<p>First, for each node \( i \) in the first hidden layer, we calculate a weighted sum \( z_i^1 \) of the input coordinates \( x_j \),</p>
|
|
|
|
$$
|
|
\begin{equation} z_i^1 = \sum_{j=1}^{M} w_{ij}^1 x_j + b_i^1
|
|
\label{_auto1}
|
|
\end{equation}
|
|
$$
|
|
|
|
<p>Here \( b_i \) is the so-called bias which is normally needed in
|
|
case of zero activation weights or inputs. How to fix the biases and
|
|
the weights will be discussed below. The value of \( z_i^1 \) is the
|
|
argument to the activation function \( f_i \) of each node \( i \), The
|
|
variable \( M \) stands for all possible inputs to a given node \( i \) in the
|
|
first layer. We define the output \( y_i^1 \) of all neurons in layer 1 as
|
|
</p>
|
|
|
|
$$
|
|
\begin{equation}
|
|
y_i^1 = f(z_i^1) = f\left(\sum_{j=1}^M w_{ij}^1 x_j + b_i^1\right)
|
|
\label{outputLayer1}
|
|
\end{equation}
|
|
$$
|
|
|
|
<p>where we assume that all nodes in the same layer have identical
|
|
activation functions, hence the notation \( f \). In general, we could assume in the more general case that different layers have different activation functions.
|
|
In this case we would identify these functions with a superscript \( l \) for the \( l \)-th layer,
|
|
</p>
|
|
|
|
$$
|
|
\begin{equation}
|
|
y_i^l = f^l(u_i^l) = f^l\left(\sum_{j=1}^{N_{l-1}} w_{ij}^l y_j^{l-1} + b_i^l\right)
|
|
\label{generalLayer}
|
|
\end{equation}
|
|
$$
|
|
|
|
<p>where \( N_l \) is the number of nodes in layer \( l \). When the output of
|
|
all the nodes in the first hidden layer are computed, the values of
|
|
the subsequent layer can be calculated and so forth until the output
|
|
is obtained.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="mathematical-model">Mathematical model </h2>
|
|
|
|
<p>The output of neuron \( i \) in layer 2 is thus,</p>
|
|
|
|
$$
|
|
\begin{align}
|
|
y_i^2 &= f^2\left(\sum_{j=1}^N w_{ij}^2 y_j^1 + b_i^2\right)
|
|
\label{_auto2}\\
|
|
&= f^2\left[\sum_{j=1}^N w_{ij}^2f^1\left(\sum_{k=1}^M w_{jk}^1 x_k + b_j^1\right) + b_i^2\right]
|
|
\label{outputLayer2}
|
|
\end{align}
|
|
$$
|
|
|
|
<p>where we have substituted \( y_k^1 \) with the inputs \( x_k \). Finally, the ANN output reads</p>
|
|
|
|
$$
|
|
\begin{align}
|
|
y_i^3 &= f^3\left(\sum_{j=1}^N w_{ij}^3 y_j^2 + b_i^3\right)
|
|
\label{_auto3}\\
|
|
&= f_3\left[\sum_{j} w_{ij}^3 f^2\left(\sum_{k} w_{jk}^2 f^1\left(\sum_{m} w_{km}^1 x_m + b_k^1\right) + b_j^2\right)
|
|
+ b_1^3\right]
|
|
\label{_auto4}
|
|
\end{align}
|
|
$$
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="mathematical-model">Mathematical model </h2>
|
|
|
|
<p>We can generalize this expression to an MLP with \( l \) hidden
|
|
layers. The complete functional form is,
|
|
</p>
|
|
|
|
$$
|
|
\begin{align}
|
|
&y^{l+1}_i = f^{l+1}\left[\!\sum_{j=1}^{N_l} w_{ij}^3 f^l\left(\sum_{k=1}^{N_{l-1}}w_{jk}^{l-1}\left(\dots f^1\left(\sum_{n=1}^{N_0} w_{mn}^1 x_n+ b_m^1\right)\dots\right)+b_k^2\right)+b_1^3\right] &&
|
|
\label{completeNN}
|
|
\end{align}
|
|
$$
|
|
|
|
<p>which illustrates a basic property of MLPs: The only independent
|
|
variables are the input values \( x_n \).
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h2 id="mathematical-model">Mathematical model </h2>
|
|
|
|
<p>This confirms that an MLP, despite its quite convoluted mathematical
|
|
form, is nothing more than an analytic function, specifically a
|
|
mapping of real-valued vectors \( \hat{x} \in \mathbb{R}^n \rightarrow
|
|
\hat{y} \in \mathbb{R}^m \).
|
|
</p>
|
|
|
|
<p>Furthermore, the flexibility and universality of an MLP can be
|
|
illustrated by realizing that the expression is essentially a nested
|
|
sum of scaled activation functions of the form
|
|
</p>
|
|
|
|
$$
|
|
\begin{equation}
|
|
f(x) = c_1 f(c_2 x + c_3) + c_4
|
|
\label{_auto5}
|
|
\end{equation}
|
|
$$
|
|
|
|
<p>where the parameters \( c_i \) are weights and biases. By adjusting these
|
|
parameters, the activation functions can be shifted up and down or
|
|
left and right, change slope or be rescaled which is the key to the
|
|
flexibility of a neural network.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h3 id="matrix-vector-notation">Matrix-vector notation </h3>
|
|
|
|
<p>We can introduce a more convenient notation for the activations in an A NN. </p>
|
|
|
|
<p>Additionally, we can represent the biases and activations
|
|
as layer-wise column vectors \( \hat{b}_l \) and \( \hat{y}_l \), so that the \( i \)-th element of each vector
|
|
is the bias \( b_i^l \) and activation \( y_i^l \) of node \( i \) in layer \( l \) respectively.
|
|
</p>
|
|
|
|
<p>We have that \( \mathrm{W}_l \) is an \( N_{l-1} \times N_l \) matrix, while \( \hat{b}_l \) and \( \hat{y}_l \) are \( N_l \times 1 \) column vectors.
|
|
With this notation, the sum becomes a matrix-vector multiplication, and we can write
|
|
the equation for the activations of hidden layer 2 (assuming three nodes for simplicity) as
|
|
</p>
|
|
$$
|
|
\begin{equation}
|
|
\hat{y}_2 = f_2(\mathrm{W}_2 \hat{y}_{1} + \hat{b}_{2}) =
|
|
f_2\left(\left[\begin{array}{ccc}
|
|
w^2_{11} &w^2_{12} &w^2_{13} \\
|
|
w^2_{21} &w^2_{22} &w^2_{23} \\
|
|
w^2_{31} &w^2_{32} &w^2_{33} \\
|
|
\end{array} \right] \cdot
|
|
\left[\begin{array}{c}
|
|
y^1_1 \\
|
|
y^1_2 \\
|
|
y^1_3 \\
|
|
\end{array}\right] +
|
|
\left[\begin{array}{c}
|
|
b^2_1 \\
|
|
b^2_2 \\
|
|
b^2_3 \\
|
|
\end{array}\right]\right).
|
|
\label{_auto6}
|
|
\end{equation}
|
|
$$
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h3 id="matrix-vector-notation-and-activation">Matrix-vector notation and activation </h3>
|
|
|
|
<p>The activation of node \( i \) in layer 2 is</p>
|
|
|
|
$$
|
|
\begin{equation}
|
|
y^2_i = f_2\Bigr(w^2_{i1}y^1_1 + w^2_{i2}y^1_2 + w^2_{i3}y^1_3 + b^2_i\Bigr) =
|
|
f_2\left(\sum_{j=1}^3 w^2_{ij} y_j^1 + b^2_i\right).
|
|
\label{_auto7}
|
|
\end{equation}
|
|
$$
|
|
|
|
<p>This is not just a convenient and compact notation, but also a useful
|
|
and intuitive way to think about MLPs: The output is calculated by a
|
|
series of matrix-vector multiplications and vector additions that are
|
|
used as input to the activation functions. For each operation
|
|
\( \mathrm{W}_l \hat{y}_{l-1} \) we move forward one layer.
|
|
</p>
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h3 id="activation-functions">Activation functions </h3>
|
|
|
|
<p>A property that characterizes a neural network, other than its
|
|
connectivity, is the choice of activation function(s). As described
|
|
in, the following restrictions are imposed on an activation function
|
|
for a FFNN to fulfill the universal approximation theorem
|
|
</p>
|
|
|
|
<ul>
|
|
<li> Non-constant</li>
|
|
<li> Bounded</li>
|
|
<li> Monotonically-increasing</li>
|
|
<li> Continuous</li>
|
|
</ul>
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h3 id="activation-functions-logistic-and-hyperbolic-ones">Activation functions, Logistic and Hyperbolic ones </h3>
|
|
|
|
<p>The second requirement excludes all linear functions. Furthermore, in
|
|
a MLP with only linear activation functions, each layer simply
|
|
performs a linear transformation of its inputs.
|
|
</p>
|
|
|
|
<p>Regardless of the number of layers, the output of the NN will be
|
|
nothing but a linear function of the inputs. Thus we need to introduce
|
|
some kind of non-linearity to the NN to be able to fit non-linear
|
|
functions Typical examples are the logistic <em>Sigmoid</em>
|
|
</p>
|
|
|
|
$$
|
|
f(x) = \frac{1}{1 + e^{-x}},
|
|
$$
|
|
|
|
<p>and the <em>hyperbolic tangent</em> function</p>
|
|
$$
|
|
f(x) = \tanh(x)
|
|
$$
|
|
|
|
|
|
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
|
<h3 id="relevance">Relevance </h3>
|
|
|
|
<p>The <em>sigmoid</em> function are more biologically plausible because the
|
|
output of inactive neurons are zero. Such activation function are
|
|
called <em>one-sided</em>. However, it has been shown that the hyperbolic
|
|
tangent performs better than the sigmoid for training MLPs. has
|
|
become the most popular for <em>deep neural networks</em>
|
|
</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #BA2121; font-style: italic">"""The sigmoid function (or the logistic curve) is a </span>
|
|
<span style="color: #BA2121; font-style: italic">function that takes any real number, z, and outputs a number (0,1).</span>
|
|
<span style="color: #BA2121; font-style: italic">It is useful in neural networks for assigning weights on a relative scale.</span>
|
|
<span style="color: #BA2121; font-style: italic">The value z is the weighted sum of parameters involved in the learning algorithm."""</span>
|
|
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">math</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">mt</span>
|
|
|
|
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-5</span>, <span style="color: #666666">5</span>, <span style="color: #666666">.1</span>)
|
|
sigma_fn <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>vectorize(<span style="color: #008000; font-weight: bold">lambda</span> z: <span style="color: #666666">1/</span>(<span style="color: #666666">1+</span>numpy<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>z)))
|
|
sigma <span style="color: #666666">=</span> sigma_fn(z)
|
|
|
|
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
|
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
|
ax<span style="color: #666666">.</span>plot(z, sigma)
|
|
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-0.1</span>, <span style="color: #666666">1.1</span>])
|
|
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-5</span>,<span style="color: #666666">5</span>])
|
|
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
|
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
|
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'sigmoid function'</span>)
|
|
|
|
plt<span style="color: #666666">.</span>show()
|
|
|
|
<span style="color: #BA2121; font-style: italic">"""Step Function"""</span>
|
|
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-5</span>, <span style="color: #666666">5</span>, <span style="color: #666666">.02</span>)
|
|
step_fn <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>vectorize(<span style="color: #008000; font-weight: bold">lambda</span> z: <span style="color: #666666">1.0</span> <span style="color: #008000; font-weight: bold">if</span> z <span style="color: #666666">>=</span> <span style="color: #666666">0.0</span> <span style="color: #008000; font-weight: bold">else</span> <span style="color: #666666">0.0</span>)
|
|
step <span style="color: #666666">=</span> step_fn(z)
|
|
|
|
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
|
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
|
ax<span style="color: #666666">.</span>plot(z, step)
|
|
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-0.5</span>, <span style="color: #666666">1.5</span>])
|
|
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-5</span>,<span style="color: #666666">5</span>])
|
|
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
|
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
|
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'step function'</span>)
|
|
|
|
plt<span style="color: #666666">.</span>show()
|
|
|
|
<span style="color: #BA2121; font-style: italic">"""Sine Function"""</span>
|
|
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-2*</span>mt<span style="color: #666666">.</span>pi, <span style="color: #666666">2*</span>mt<span style="color: #666666">.</span>pi, <span style="color: #666666">0.1</span>)
|
|
t <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>sin(z)
|
|
|
|
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
|
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
|
ax<span style="color: #666666">.</span>plot(z, t)
|
|
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-1.0</span>, <span style="color: #666666">1.0</span>])
|
|
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-2*</span>mt<span style="color: #666666">.</span>pi,<span style="color: #666666">2*</span>mt<span style="color: #666666">.</span>pi])
|
|
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
|
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
|
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'sine function'</span>)
|
|
|
|
plt<span style="color: #666666">.</span>show()
|
|
|
|
<span style="color: #BA2121; font-style: italic">"""Plots a graph of the squashing function used by a rectified linear</span>
|
|
<span style="color: #BA2121; font-style: italic">unit"""</span>
|
|
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-2</span>, <span style="color: #666666">2</span>, <span style="color: #666666">.1</span>)
|
|
zero <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>zeros(<span style="color: #008000">len</span>(z))
|
|
y <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>max([zero, z], axis<span style="color: #666666">=0</span>)
|
|
|
|
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
|
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
|
ax<span style="color: #666666">.</span>plot(z, y)
|
|
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-2.0</span>, <span style="color: #666666">2.0</span>])
|
|
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-2.0</span>, <span style="color: #666666">2.0</span>])
|
|
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
|
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
|
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'Rectified linear unit'</span>)
|
|
|
|
plt<span style="color: #666666">.</span>show()
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
|
|
<!-- ------------------- end of main content --------------- -->
|
|
<center style="font-size:80%">
|
|
<!-- copyright --> © 1999-2025, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
|
|
</center>
|
|
</body>
|
|
</html>
|
|
|