Files
FYS-STK4155/doc/pub/week41/html/._week41-bs042.html
T
2022-10-20 08:20:59 +02:00

352 lines
20 KiB
HTML

<!--
HTML file automatically generated from DocOnce source
(https://github.com/doconce/doconce/)
doconce format html week41.do.txt --html_style=bootstrap --pygments_html_style=default --html_admon=bootstrap_panel --html_output=week41-bs --no_mako
-->
<html>
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="DocOnce: https://github.com/doconce/doconce/" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<meta name="description" content="Week 41 Constructing a Neural Network code, Tensor flow and start Convolutional Neural Networks">
<title>Week 41 Constructing a Neural Network code, Tensor flow and start Convolutional Neural Networks</title>
<!-- Bootstrap style: bootstrap -->
<!-- doconce format html week41.do.txt --html_style=bootstrap --pygments_html_style=default --html_admon=bootstrap_panel --html_output=week41-bs --no_mako -->
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
<!-- not necessary
<link href="https://netdna.bootstrapcdn.com/font-awesome/4.0.3/css/font-awesome.css" rel="stylesheet">
-->
<style type="text/css">
/* Add scrollbar to dropdown menus in bootstrap navigation bar */
.dropdown-menu {
height: auto;
max-height: 400px;
overflow-x: hidden;
}
/* Adds an invisible element before each target to offset for the navigation
bar */
.anchor::before {
content:"";
display:block;
height:50px; /* fixed header height for style bootstrap */
margin:-50px 0 0; /* negative fixed header height */
}
</style>
</head>
<!-- tocinfo
{'highest level': 2,
'sections': [('Plan for week 41', 2, None, 'plan-for-week-41'),
('Videos on Neural Networks',
2,
None,
'videos-on-neural-networks'),
('Review of the back propagation algorithm',
2,
None,
'review-of-the-back-propagation-algorithm'),
('Setting up the Back propagation algorithm',
2,
None,
'setting-up-the-back-propagation-algorithm'),
('Setting up a Multi-layer perceptron model for classification',
2,
None,
'setting-up-a-multi-layer-perceptron-model-for-classification'),
('Defining the cost function',
2,
None,
'defining-the-cost-function'),
('Example: binary classification problem',
2,
None,
'example-binary-classification-problem'),
('The Softmax function', 2, None, 'the-softmax-function'),
('Developing a code for doing neural networks with back '
'propagation',
2,
None,
'developing-a-code-for-doing-neural-networks-with-back-propagation'),
('Collect and pre-process data',
2,
None,
'collect-and-pre-process-data'),
('Train and test datasets', 2, None, 'train-and-test-datasets'),
('Define model and architecture',
2,
None,
'define-model-and-architecture'),
('Layers', 2, None, 'layers'),
('Weights and biases', 2, None, 'weights-and-biases'),
('Feed-forward pass', 2, None, 'feed-forward-pass'),
('Matrix multiplications', 2, None, 'matrix-multiplications'),
('Choose cost function and optimizer',
2,
None,
'choose-cost-function-and-optimizer'),
('Optimizing the cost function',
2,
None,
'optimizing-the-cost-function'),
('Regularization', 2, None, 'regularization'),
('Matrix multiplication', 2, None, 'matrix-multiplication'),
('Improving performance', 2, None, 'improving-performance'),
('Full object-oriented implementation',
2,
None,
'full-object-oriented-implementation'),
('Evaluate model performance on test data',
2,
None,
'evaluate-model-performance-on-test-data'),
('Adjust hyperparameters', 2, None, 'adjust-hyperparameters'),
('Visualization', 2, None, 'visualization'),
('scikit-learn implementation',
2,
None,
'scikit-learn-implementation'),
('Visualization', 2, None, 'visualization'),
('Testing our code for the XOR, OR and AND gates',
2,
None,
'testing-our-code-for-the-xor-or-and-and-gates'),
('The AND and XOR Gates', 2, None, 'the-and-and-xor-gates'),
('Representing the Data Sets',
2,
None,
'representing-the-data-sets'),
('Setting up the Neural Network',
2,
None,
'setting-up-the-neural-network'),
('The Code using Scikit-Learn',
2,
None,
'the-code-using-scikit-learn'),
('Building neural networks in Tensorflow and Keras',
2,
None,
'building-neural-networks-in-tensorflow-and-keras'),
('Tensorflow', 2, None, 'tensorflow'),
('Using Keras', 2, None, 'using-keras'),
('Collect and pre-process data',
2,
None,
'collect-and-pre-process-data'),
('The Breast Cancer Data, now with Keras',
2,
None,
'the-breast-cancer-data-now-with-keras'),
('Fine-tuning neural network hyperparameters',
2,
None,
'fine-tuning-neural-network-hyperparameters'),
('Hidden layers', 2, None, 'hidden-layers'),
('Which activation function should I use?',
2,
None,
'which-activation-function-should-i-use'),
('Is the Logistic activation function (Sigmoid) our choice?',
2,
None,
'is-the-logistic-activation-function-sigmoid-our-choice'),
('The derivative of the Logistic funtion',
2,
None,
'the-derivative-of-the-logistic-funtion'),
('The RELU function family', 2, None, 'the-relu-function-family'),
('Which activation function should we use?',
2,
None,
'which-activation-function-should-we-use'),
('More on activation functions, output layers',
2,
None,
'more-on-activation-functions-output-layers'),
('Batch Normalization', 2, None, 'batch-normalization'),
('Dropout', 2, None, 'dropout'),
('Gradient Clipping', 2, None, 'gradient-clipping'),
('A very nice website on Neural Networks',
2,
None,
'a-very-nice-website-on-neural-networks'),
('A top-down perspective on Neural networks',
2,
None,
'a-top-down-perspective-on-neural-networks'),
('Limitations of supervised learning with deep networks',
2,
None,
'limitations-of-supervised-learning-with-deep-networks')]}
end of tocinfo -->
<body>
<script type="text/x-mathjax-config">
MathJax.Hub.Config({
TeX: {
equationNumbers: { autoNumber: "none" },
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
}
});
</script>
<script type="text/javascript" async
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
</script>
<!-- Bootstrap navigation bar -->
<div class="navbar navbar-default navbar-fixed-top">
<div class="navbar-header">
<button type="button" class="navbar-toggle" data-toggle="collapse" data-target=".navbar-responsive-collapse">
<span class="icon-bar"></span>
<span class="icon-bar"></span>
<span class="icon-bar"></span>
</button>
<a class="navbar-brand" href="week41-bs.html">Week 41 Constructing a Neural Network code, Tensor flow and start Convolutional Neural Networks</a>
</div>
<div class="navbar-collapse collapse navbar-responsive-collapse">
<ul class="nav navbar-nav navbar-right">
<li class="dropdown">
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
<ul class="dropdown-menu">
<!-- navigation toc: --> <li><a href="._week41-bs001.html#plan-for-week-41" style="font-size: 80%;">Plan for week 41</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs002.html#videos-on-neural-networks" style="font-size: 80%;">Videos on Neural Networks</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs003.html#review-of-the-back-propagation-algorithm" style="font-size: 80%;">Review of the back propagation algorithm</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs004.html#setting-up-the-back-propagation-algorithm" style="font-size: 80%;">Setting up the Back propagation algorithm</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs005.html#setting-up-a-multi-layer-perceptron-model-for-classification" style="font-size: 80%;">Setting up a Multi-layer perceptron model for classification</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs006.html#defining-the-cost-function" style="font-size: 80%;">Defining the cost function</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs007.html#example-binary-classification-problem" style="font-size: 80%;">Example: binary classification problem</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs008.html#the-softmax-function" style="font-size: 80%;">The Softmax function</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs009.html#developing-a-code-for-doing-neural-networks-with-back-propagation" style="font-size: 80%;">Developing a code for doing neural networks with back propagation</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs036.html#collect-and-pre-process-data" style="font-size: 80%;">Collect and pre-process data</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs011.html#train-and-test-datasets" style="font-size: 80%;">Train and test datasets</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs012.html#define-model-and-architecture" style="font-size: 80%;">Define model and architecture</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs013.html#layers" style="font-size: 80%;">Layers</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs014.html#weights-and-biases" style="font-size: 80%;">Weights and biases</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs015.html#feed-forward-pass" style="font-size: 80%;">Feed-forward pass</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs016.html#matrix-multiplications" style="font-size: 80%;">Matrix multiplications</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs017.html#choose-cost-function-and-optimizer" style="font-size: 80%;">Choose cost function and optimizer</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs018.html#optimizing-the-cost-function" style="font-size: 80%;">Optimizing the cost function</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs019.html#regularization" style="font-size: 80%;">Regularization</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs020.html#matrix-multiplication" style="font-size: 80%;">Matrix multiplication</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs021.html#improving-performance" style="font-size: 80%;">Improving performance</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs022.html#full-object-oriented-implementation" style="font-size: 80%;">Full object-oriented implementation</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs023.html#evaluate-model-performance-on-test-data" style="font-size: 80%;">Evaluate model performance on test data</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs024.html#adjust-hyperparameters" style="font-size: 80%;">Adjust hyperparameters</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs027.html#visualization" style="font-size: 80%;">Visualization</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs026.html#scikit-learn-implementation" style="font-size: 80%;">scikit-learn implementation</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs027.html#visualization" style="font-size: 80%;">Visualization</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs028.html#testing-our-code-for-the-xor-or-and-and-gates" style="font-size: 80%;">Testing our code for the XOR, OR and AND gates</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs029.html#the-and-and-xor-gates" style="font-size: 80%;">The AND and XOR Gates</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs030.html#representing-the-data-sets" style="font-size: 80%;">Representing the Data Sets</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs031.html#setting-up-the-neural-network" style="font-size: 80%;">Setting up the Neural Network</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs032.html#the-code-using-scikit-learn" style="font-size: 80%;">The Code using Scikit-Learn</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs033.html#building-neural-networks-in-tensorflow-and-keras" style="font-size: 80%;">Building neural networks in Tensorflow and Keras</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs034.html#tensorflow" style="font-size: 80%;">Tensorflow</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs035.html#using-keras" style="font-size: 80%;">Using Keras</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs036.html#collect-and-pre-process-data" style="font-size: 80%;">Collect and pre-process data</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs037.html#the-breast-cancer-data-now-with-keras" style="font-size: 80%;">The Breast Cancer Data, now with Keras</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs038.html#fine-tuning-neural-network-hyperparameters" style="font-size: 80%;">Fine-tuning neural network hyperparameters</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs039.html#hidden-layers" style="font-size: 80%;">Hidden layers</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs040.html#which-activation-function-should-i-use" style="font-size: 80%;">Which activation function should I use?</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs041.html#is-the-logistic-activation-function-sigmoid-our-choice" style="font-size: 80%;">Is the Logistic activation function (Sigmoid) our choice?</a></li>
<!-- navigation toc: --> <li><a href="#the-derivative-of-the-logistic-funtion" style="font-size: 80%;">The derivative of the Logistic funtion</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs043.html#the-relu-function-family" style="font-size: 80%;">The RELU function family</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs044.html#which-activation-function-should-we-use" style="font-size: 80%;">Which activation function should we use?</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs045.html#more-on-activation-functions-output-layers" style="font-size: 80%;">More on activation functions, output layers</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs046.html#batch-normalization" style="font-size: 80%;">Batch Normalization</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs047.html#dropout" style="font-size: 80%;">Dropout</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs048.html#gradient-clipping" style="font-size: 80%;">Gradient Clipping</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs049.html#a-very-nice-website-on-neural-networks" style="font-size: 80%;">A very nice website on Neural Networks</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs050.html#a-top-down-perspective-on-neural-networks" style="font-size: 80%;">A top-down perspective on Neural networks</a></li>
<!-- navigation toc: --> <li><a href="._week41-bs051.html#limitations-of-supervised-learning-with-deep-networks" style="font-size: 80%;">Limitations of supervised learning with deep networks</a></li>
</ul>
</li>
</ul>
</div>
</div>
</div> <!-- end of navigation bar -->
<div class="container">
<p>&nbsp;</p><p>&nbsp;</p><p>&nbsp;</p> <!-- add vertical space -->
<a name="part0042"></a>
<!-- !split -->
<h2 id="the-derivative-of-the-logistic-funtion" class="anchor">The derivative of the Logistic funtion </h2>
<p>Looking at the logistic activation function, when inputs become large
(negative or positive), the function saturates at 0 or 1, with a
derivative extremely close to 0. Thus when backpropagation kicks in,
it has virtually no gradient to propagate back through the network,
and what little gradient exists keeps getting diluted as
backpropagation progresses down through the top layers, so there is
really nothing left for the lower layers.
</p>
<p>In their paper, Glorot and Bengio propose a way to significantly
alleviate this problem. We need the signal to flow properly in both
directions: in the forward direction when making predictions, and in
the reverse direction when backpropagating gradients. We don&#8217;t want
the signal to die out, nor do we want it to explode and saturate. For
the signal to flow properly, the authors argue that we need the
variance of the outputs of each layer to be equal to the variance of
its inputs, and we also need the gradients to have equal variance
before and after flowing through a layer in the reverse direction.
</p>
<p>One of the insights in the 2010 paper by Glorot and Bengio was that
the vanishing/exploding gradients problems were in part due to a poor
choice of activation function. Until then most people had assumed that
if Nature had chosen to use roughly sigmoid activation functions in
biological neurons, they must be an excellent choice. But it turns out
that other activation functions behave much better in deep neural
networks, in particular the ReLU activation function, mostly because
it does not saturate for positive values (and also because it is quite
fast to compute).
</p>
<p>
<!-- navigation buttons at the bottom of the page -->
<ul class="pagination">
<li><a href="._week41-bs041.html">&laquo;</a></li>
<li><a href="._week41-bs000.html">1</a></li>
<li><a href="">...</a></li>
<li><a href="._week41-bs034.html">35</a></li>
<li><a href="._week41-bs035.html">36</a></li>
<li><a href="._week41-bs036.html">37</a></li>
<li><a href="._week41-bs037.html">38</a></li>
<li><a href="._week41-bs038.html">39</a></li>
<li><a href="._week41-bs039.html">40</a></li>
<li><a href="._week41-bs040.html">41</a></li>
<li><a href="._week41-bs041.html">42</a></li>
<li class="active"><a href="._week41-bs042.html">43</a></li>
<li><a href="._week41-bs043.html">44</a></li>
<li><a href="._week41-bs044.html">45</a></li>
<li><a href="._week41-bs045.html">46</a></li>
<li><a href="._week41-bs046.html">47</a></li>
<li><a href="._week41-bs047.html">48</a></li>
<li><a href="._week41-bs048.html">49</a></li>
<li><a href="._week41-bs049.html">50</a></li>
<li><a href="._week41-bs050.html">51</a></li>
<li><a href="._week41-bs051.html">52</a></li>
<li><a href="._week41-bs043.html">&raquo;</a></li>
</ul>
<!-- ------------------- end of main content --------------- -->
</div> <!-- end container -->
<!-- include javascript, jQuery *first* -->
<script src="https://ajax.googleapis.com/ajax/libs/jquery/1.10.2/jquery.min.js"></script>
<script src="https://netdna.bootstrapcdn.com/bootstrap/3.0.0/js/bootstrap.min.js"></script>
<!-- Bootstrap footer
<footer>
<a href="https://..."><img width="250" align=right src="https://..."></a>
</footer>
-->
<center style="font-size:80%">
<!-- copyright only on the titlepage -->
</center>
</body>
</html>