From b09ab9c83c998fa5f6502d7a56460b8d78705ad8 Mon Sep 17 00:00:00 2001 From: mhjensen Date: Fri, 9 Oct 2020 06:04:27 +0200 Subject: [PATCH] updating week41 --- doc/pub/week41/html/._week41-bs000.html | 117 ++- doc/pub/week41/html/._week41-bs001.html | 115 ++- doc/pub/week41/html/._week41-bs002.html | 115 ++- doc/pub/week41/html/._week41-bs003.html | 115 ++- doc/pub/week41/html/._week41-bs004.html | 115 ++- doc/pub/week41/html/._week41-bs005.html | 115 ++- doc/pub/week41/html/._week41-bs006.html | 115 ++- doc/pub/week41/html/._week41-bs007.html | 115 ++- doc/pub/week41/html/._week41-bs008.html | 115 ++- doc/pub/week41/html/._week41-bs009.html | 115 ++- doc/pub/week41/html/._week41-bs010.html | 115 ++- doc/pub/week41/html/._week41-bs011.html | 115 ++- doc/pub/week41/html/._week41-bs012.html | 115 ++- doc/pub/week41/html/._week41-bs013.html | 115 ++- doc/pub/week41/html/._week41-bs014.html | 115 ++- doc/pub/week41/html/._week41-bs015.html | 115 ++- doc/pub/week41/html/._week41-bs016.html | 115 ++- doc/pub/week41/html/._week41-bs017.html | 115 ++- doc/pub/week41/html/._week41-bs018.html | 115 ++- doc/pub/week41/html/._week41-bs019.html | 115 ++- doc/pub/week41/html/._week41-bs020.html | 115 ++- doc/pub/week41/html/._week41-bs021.html | 115 ++- doc/pub/week41/html/._week41-bs022.html | 115 ++- doc/pub/week41/html/._week41-bs023.html | 115 ++- doc/pub/week41/html/._week41-bs024.html | 115 ++- doc/pub/week41/html/._week41-bs025.html | 115 ++- doc/pub/week41/html/._week41-bs026.html | 115 ++- doc/pub/week41/html/._week41-bs027.html | 127 ++- doc/pub/week41/html/._week41-bs028.html | 187 ++--- doc/pub/week41/html/._week41-bs029.html | 374 +++++---- doc/pub/week41/html/._week41-bs030.html | 331 +++++--- doc/pub/week41/html/._week41-bs031.html | 227 +++--- doc/pub/week41/html/._week41-bs032.html | 295 +++---- doc/pub/week41/html/._week41-bs033.html | 115 ++- doc/pub/week41/html/._week41-bs034.html | 147 ++-- doc/pub/week41/html/._week41-bs035.html | 163 ++-- doc/pub/week41/html/._week41-bs036.html | 158 ++-- doc/pub/week41/html/._week41-bs037.html | 147 ++-- doc/pub/week41/html/._week41-bs038.html | 165 ++-- doc/pub/week41/html/._week41-bs039.html | 147 ++-- doc/pub/week41/html/._week41-bs040.html | 157 ++-- doc/pub/week41/html/._week41-bs041.html | 145 ++-- doc/pub/week41/html/._week41-bs042.html | 189 +++-- doc/pub/week41/html/._week41-bs043.html | 153 ++-- doc/pub/week41/html/._week41-bs044.html | 165 ++-- doc/pub/week41/html/._week41-bs045.html | 161 ++-- doc/pub/week41/html/._week41-bs046.html | 267 +++--- doc/pub/week41/html/._week41-bs047.html | 267 +++--- doc/pub/week41/html/._week41-bs048.html | 263 +++--- doc/pub/week41/html/._week41-bs049.html | 267 +++--- doc/pub/week41/html/._week41-bs050.html | 266 +++--- doc/pub/week41/html/._week41-bs051.html | 257 +++--- doc/pub/week41/html/._week41-bs052.html | 295 ++++--- doc/pub/week41/html/._week41-bs053.html | 267 +++--- doc/pub/week41/html/._week41-bs054.html | 399 ++++----- doc/pub/week41/html/._week41-bs055.html | 279 ++++--- doc/pub/week41/html/._week41-bs056.html | 300 +++---- doc/pub/week41/html/._week41-bs057.html | 288 ++++--- doc/pub/week41/html/._week41-bs058.html | 293 ++++--- doc/pub/week41/html/._week41-bs059.html | 289 ++++--- doc/pub/week41/html/._week41-bs060.html | 281 ++++--- doc/pub/week41/html/week41-bs.html | 117 ++- doc/pub/week41/html/week41-reveal.html | 749 ++++++++++------- doc/pub/week41/html/week41-solarized.html | 798 +++++++++++------- doc/pub/week41/html/week41.html | 798 +++++++++++------- doc/pub/week41/ipynb/ipynb-week41-src.tar.gz | Bin 87344 -> 87344 bytes doc/pub/week41/ipynb/week41.ipynb | 804 +++++++++++-------- doc/src/week41/week41.do.txt | 624 ++++++++------ 68 files changed, 8896 insertions(+), 5932 deletions(-) diff --git a/doc/pub/week41/html/._week41-bs000.html b/doc/pub/week41/html/._week41-bs000.html index 045b39a7d..6c79be947 100644 --- a/doc/pub/week41/html/._week41-bs000.html +++ b/doc/pub/week41/html/._week41-bs000.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -227,7 +272,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 8, 2020

    +

    Oct 9, 2020


    @@ -251,7 +296,7 @@ MathJax.Hub.Config({

  • 9
  • 10
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs001.html b/doc/pub/week41/html/._week41-bs001.html index 6dfba7166..2fcb5d6bd 100644 --- a/doc/pub/week41/html/._week41-bs001.html +++ b/doc/pub/week41/html/._week41-bs001.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -235,7 +280,7 @@ extbooks/TensorflowML.pdf" target="_self">Aurelien Geron's chapters 10-11 an
  • 10
  • 11
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs002.html b/doc/pub/week41/html/._week41-bs002.html index 46056d075..77bb370b2 100644 --- a/doc/pub/week41/html/._week41-bs002.html +++ b/doc/pub/week41/html/._week41-bs002.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -299,7 +344,7 @@ Here it is convenient to use stochastic gradient descent (see the examples below
  • 11
  • 12
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs003.html b/doc/pub/week41/html/._week41-bs003.html index 15581face..96a811800 100644 --- a/doc/pub/week41/html/._week41-bs003.html +++ b/doc/pub/week41/html/._week41-bs003.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -262,7 +307,7 @@ of our network.
  • 12
  • 13
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs004.html b/doc/pub/week41/html/._week41-bs004.html index 2d14dd633..a4af2b6fe 100644 --- a/doc/pub/week41/html/._week41-bs004.html +++ b/doc/pub/week41/html/._week41-bs004.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -285,7 +330,7 @@ The back propagation equations need now only a small change, namely the definiti
  • 13
  • 14
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs005.html b/doc/pub/week41/html/._week41-bs005.html index f1f8cb4f1..aac5db7df 100644 --- a/doc/pub/week41/html/._week41-bs005.html +++ b/doc/pub/week41/html/._week41-bs005.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -274,7 +319,7 @@ In case we use another activation function than the logistic one, we need to eva
  • 14
  • 15
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs006.html b/doc/pub/week41/html/._week41-bs006.html index 6e47b920a..e54c4f3af 100644 --- a/doc/pub/week41/html/._week41-bs006.html +++ b/doc/pub/week41/html/._week41-bs006.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -249,7 +294,7 @@ which in case of the simply binary model reduces to having \( i=j \).
  • 15
  • 16
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs007.html b/doc/pub/week41/html/._week41-bs007.html index ae77b54c1..e843854a9 100644 --- a/doc/pub/week41/html/._week41-bs007.html +++ b/doc/pub/week41/html/._week41-bs007.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -244,7 +289,7 @@ One can identify a set of key steps when using neural networks to solve supervis
  • 16
  • 17
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs008.html b/doc/pub/week41/html/._week41-bs008.html index d6d3caf3d..f4b9cb615 100644 --- a/doc/pub/week41/html/._week41-bs008.html +++ b/doc/pub/week41/html/._week41-bs008.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -324,7 +369,7 @@ plt.show()
  • 17
  • 18
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs009.html b/doc/pub/week41/html/._week41-bs009.html index e972f1efa..aacbe2c20 100644 --- a/doc/pub/week41/html/._week41-bs009.html +++ b/doc/pub/week41/html/._week41-bs009.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -279,7 +324,7 @@ X_train, X_test, Y_train, Y_test = train_tes
  • 18
  • 19
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs010.html b/doc/pub/week41/html/._week41-bs010.html index 86a0fe91b..abab9395a 100644 --- a/doc/pub/week41/html/._week41-bs010.html +++ b/doc/pub/week41/html/._week41-bs010.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -277,7 +322,7 @@ which is inspired by probability theory (see logistic regression) and was most c
  • 19
  • 20
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs011.html b/doc/pub/week41/html/._week41-bs011.html index 86da1a213..0c358b77d 100644 --- a/doc/pub/week41/html/._week41-bs011.html +++ b/doc/pub/week41/html/._week41-bs011.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -276,7 +321,7 @@ weights to the output layer.
  • 20
  • 21
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs012.html b/doc/pub/week41/html/._week41-bs012.html index 11a3b0026..5a215e367 100644 --- a/doc/pub/week41/html/._week41-bs012.html +++ b/doc/pub/week41/html/._week41-bs012.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -267,7 +312,7 @@ output_bias = np21
  • 22
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs013.html b/doc/pub/week41/html/._week41-bs013.html index 95d0c5c9f..cae048755 100644 --- a/doc/pub/week41/html/._week41-bs013.html +++ b/doc/pub/week41/html/._week41-bs013.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -258,7 +303,7 @@ $$ a_{j}^{L} = \frac{\exp{(z_j^{L})}}
  • 22
  • 23
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs014.html b/doc/pub/week41/html/._week41-bs014.html index 5e0fe8f6c..dd4c8c886 100644 --- a/doc/pub/week41/html/._week41-bs014.html +++ b/doc/pub/week41/html/._week41-bs014.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -306,7 +351,7 @@ predictions = predict(X_train)
  • 23
  • 24
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs015.html b/doc/pub/week41/html/._week41-bs015.html index 593cd7abd..a3c670f31 100644 --- a/doc/pub/week41/html/._week41-bs015.html +++ b/doc/pub/week41/html/._week41-bs015.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -262,7 +307,7 @@ you got the correct label. The probability of category \( c \) is given by the s
  • 24
  • 25
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs016.html b/doc/pub/week41/html/._week41-bs016.html index a0aaead4e..9f9adffee 100644 --- a/doc/pub/week41/html/._week41-bs016.html +++ b/doc/pub/week41/html/._week41-bs016.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -271,7 +316,7 @@ The various optmization methods, with codes and algorithms, are discussed in o
  • 25
  • 26
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs017.html b/doc/pub/week41/html/._week41-bs017.html index 29fa1c793..22119ce81 100644 --- a/doc/pub/week41/html/._week41-bs017.html +++ b/doc/pub/week41/html/._week41-bs017.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -264,7 +309,7 @@ calculate the gradient efficently.
  • 26
  • 27
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs018.html b/doc/pub/week41/html/._week41-bs018.html index e5b290858..16c5a6494 100644 --- a/doc/pub/week41/html/._week41-bs018.html +++ b/doc/pub/week41/html/._week41-bs018.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -345,7 +390,7 @@ lmbd = 0.0127
  • 28
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs019.html b/doc/pub/week41/html/._week41-bs019.html index d8a1b5cb3..8e2102e49 100644 --- a/doc/pub/week41/html/._week41-bs019.html +++ b/doc/pub/week41/html/._week41-bs019.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -251,7 +296,7 @@ Andrew Ng goes through some of these considerations in this 28
  • 29
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs020.html b/doc/pub/week41/html/._week41-bs020.html index ce7cea4a6..d22f3134f 100644 --- a/doc/pub/week41/html/._week41-bs020.html +++ b/doc/pub/week41/html/._week41-bs020.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -343,7 +388,7 @@ being realizations of this object with different hyperparameters. An implementat
  • 29
  • 30
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs021.html b/doc/pub/week41/html/._week41-bs021.html index f0fa21428..6396441b7 100644 --- a/doc/pub/week41/html/._week41-bs021.html +++ b/doc/pub/week41/html/._week41-bs021.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -266,7 +311,7 @@ test_predict = dnn30
  • 31
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs022.html b/doc/pub/week41/html/._week41-bs022.html index 53f0575ac..0165691b7 100644 --- a/doc/pub/week41/html/._week41-bs022.html +++ b/doc/pub/week41/html/._week41-bs022.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -264,7 +309,7 @@ DNN_numpy = np.
  • 31
  • 32
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs023.html b/doc/pub/week41/html/._week41-bs023.html index 583224d05..3350befa4 100644 --- a/doc/pub/week41/html/._week41-bs023.html +++ b/doc/pub/week41/html/._week41-bs023.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -273,7 +318,7 @@ plt.show()
  • 32
  • 33
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs024.html b/doc/pub/week41/html/._week41-bs024.html index 1204b5d13..3e6e635f1 100644 --- a/doc/pub/week41/html/._week41-bs024.html +++ b/doc/pub/week41/html/._week41-bs024.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -270,7 +315,7 @@ DNN_scikit = np
  • 33
  • 34
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs025.html b/doc/pub/week41/html/._week41-bs025.html index 1a992141c..75e99dc80 100644 --- a/doc/pub/week41/html/._week41-bs025.html +++ b/doc/pub/week41/html/._week41-bs025.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -273,7 +318,7 @@ plt.show()
  • 34
  • 35
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs026.html b/doc/pub/week41/html/._week41-bs026.html index dde4abd53..f8b41152a 100644 --- a/doc/pub/week41/html/._week41-bs026.html +++ b/doc/pub/week41/html/._week41-bs026.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -246,7 +291,7 @@ NumPy arrays.
  • 35
  • 36
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs027.html b/doc/pub/week41/html/._week41-bs027.html index ba7e13d54..c91870cf7 100644 --- a/doc/pub/week41/html/._week41-bs027.html +++ b/doc/pub/week41/html/._week41-bs027.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -245,10 +290,20 @@ To install tensorflow on Unix/Linux systems, use pip as

    and/or if you use anaconda, just write (or install from the graphical user interface) +(current release of CPU-only TensorFlow)

    -

    conda install tensorflow
    +
    conda create -n tf tensorflow
    +conda activate tf
    +
    +

    +To install the current release of GPU TensorFlow +

    + + +

    conda create -n tf-gpu tensorflow-gpu
    +conda activate tf-gpu
     

    @@ -276,7 +331,7 @@ and/or if you use anaconda, just write (or install from the graphical use

  • 36
  • 37
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs028.html b/doc/pub/week41/html/._week41-bs028.html index 4599537ae..c333dd65e 100644 --- a/doc/pub/week41/html/._week41-bs028.html +++ b/doc/pub/week41/html/._week41-bs028.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,69 +253,29 @@ MathJax.Hub.Config({ -

    Collect and pre-process data

    +

    Using Keras

    + +

    +Keras is a high level neural network +that supports Tensorflow, CTNK and Theano as backends. +If you have Tensorflow installed Keras is available through the tf.keras module. +If you have Anaconda installed you may run the following command +

    + + +

    conda install keras
    +
    +

    +Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager:

    -

    # import necessary packages
    -import numpy as np
    -import matplotlib.pyplot as plt
    -from sklearn import datasets
    -
    -
    -# ensure the same random numbers appear every time
    -np.random.seed(0)
    -
    -# display images in notebook
    -%matplotlib inline
    -plt.rcParams['figure.figsize'] = (12,12)
    -
    -
    -# download MNIST dataset
    -digits = datasets.load_digits()
    -
    -# define inputs and labels
    -inputs = digits.images
    -labels = digits.target
    -
    -print("inputs = (n_inputs, pixel_width, pixel_height) = " + str(inputs.shape))
    -print("labels = (n_inputs) = " + str(labels.shape))
    -
    -
    -# flatten the image
    -# the value -1 means dimension is inferred from the remaining dimensions: 8x8 = 64
    -n_inputs = len(inputs)
    -inputs = inputs.reshape(n_inputs, -1)
    -print("X = (n_inputs, n_features) = " + str(inputs.shape))
    -
    -
    -# choose some random images to display
    -indices = np.arange(n_inputs)
    -random_indices = np.random.choice(indices, size=5)
    -
    -for i, image in enumerate(digits.images[random_indices]):
    -    plt.subplot(1, 5, i+1)
    -    plt.axis('off')
    -    plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest')
    -    plt.title("Label: %d" % digits.target[random_indices[i]])
    -plt.show()
    +
    pip install keras
     

    +or look up the instructions here. - -

    from keras.utils import to_categorical
    -from sklearn.model_selection import train_test_split
    -
    -# one-hot representation of labels
    -labels = to_categorical(labels)
    -
    -# split into train and test data
    -train_size = 0.8
    -test_size = 1 - train_size
    -X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,
    -                                                    test_size=test_size)
    -

    @@ -297,7 +302,7 @@ X_train, X_test, Y_train, Y_test = train_tes

  • 37
  • 38
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs029.html b/doc/pub/week41/html/._week41-bs029.html index 19dae5624..f9bfbf306 100644 --- a/doc/pub/week41/html/._week41-bs029.html +++ b/doc/pub/week41/html/._week41-bs029.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,146 +253,143 @@ MathJax.Hub.Config({ -

    Using TensorFlow backend

    - -
      -
    1. Define model and architecture
    2. -
    3. Choose cost function and optimizer
    4. -
    +

    Collect and pre-process data

    -

    import tensorflow as tf
    +
    # import necessary packages
    +import numpy as np
    +import matplotlib.pyplot as plt
    +import tensorflow as tf
    +from sklearn import datasets
     
    -class NeuralNetworkTensorflow:
    -    def __init__(
    -            self,
    -            X_train,
    -            Y_train,
    -            X_test,
    -            Y_test,
    -            n_neurons_layer1=100,
    -            n_neurons_layer2=50,
    -            n_categories=2,
    -            epochs=10,
    -            batch_size=100,
    -            eta=0.1,
    -            lmbd=0.0):
    -        
    -        # keep track of number of steps
    -        self.global_step = tf.Variable(0, dtype=tf.int32, trainable=False, name='global_step')
    -        
    -        self.X_train = X_train
    -        self.Y_train = Y_train
    -        self.X_test = X_test
    -        self.Y_test = Y_test
    -        
    -        self.n_inputs = X_train.shape[0]
    -        self.n_features = X_train.shape[1]
    -        self.n_neurons_layer1 = n_neurons_layer1
    -        self.n_neurons_layer2 = n_neurons_layer2
    -        self.n_categories = n_categories
    -        
    -        self.epochs = epochs
    -        self.batch_size = batch_size
    -        self.iterations = self.n_inputs // self.batch_size
    -        self.eta = eta
    -        self.lmbd = lmbd
    -        
    -        # build network piece by piece
    -        # name scopes (with) are used to enforce creation of new variables
    -        # https://www.tensorflow.org/guide/variables
    -        self.create_placeholders()
    -        self.create_DNN()
    -        self.create_loss()
    -        self.create_optimiser()
    -        self.create_accuracy()
    -    
    -    def create_placeholders(self):
    -        # placeholders are fine here, but "Datasets" are the preferred method
    -        # of streaming data into a model
    -        with tf.name_scope('data'):
    -            self.X = tf.placeholder(tf.float32, shape=(None, self.n_features), name='X_data')
    -            self.Y = tf.placeholder(tf.float32, shape=(None, self.n_categories), name='Y_data')
    -    
    -    def create_DNN(self):
    -        with tf.name_scope('DNN'):
    -            # the weights are stored to calculate regularization loss later
    -            
    -            # Fully connected layer 1
    -            self.W_fc1 = self.weight_variable([self.n_features, self.n_neurons_layer1], name='fc1', dtype=tf.float32)
    -            b_fc1 = self.bias_variable([self.n_neurons_layer1], name='fc1', dtype=tf.float32)
    -            a_fc1 = tf.nn.sigmoid(tf.matmul(self.X, self.W_fc1) + b_fc1)
    -            
    -            # Fully connected layer 2
    -            self.W_fc2 = self.weight_variable([self.n_neurons_layer1, self.n_neurons_layer2], name='fc2', dtype=tf.float32)
    -            b_fc2 = self.bias_variable([self.n_neurons_layer2], name='fc2', dtype=tf.float32)
    -            a_fc2 = tf.nn.sigmoid(tf.matmul(a_fc1, self.W_fc2) + b_fc2)
    -            
    -            # Output layer
    -            self.W_out = self.weight_variable([self.n_neurons_layer2, self.n_categories], name='out', dtype=tf.float32)
    -            b_out = self.bias_variable([self.n_categories], name='out', dtype=tf.float32)
    -            self.z_out = tf.matmul(a_fc2, self.W_out) + b_out
    -    
    -    def create_loss(self):
    -        with tf.name_scope('loss'):
    -            softmax_loss = tf.reduce_mean(tf.nn.softmax_cross_entropy_with_logits_v2(labels=self.Y, logits=self.z_out))
    -            
    -            regularizer_loss_fc1 = tf.nn.l2_loss(self.W_fc1)
    -            regularizer_loss_fc2 = tf.nn.l2_loss(self.W_fc2)
    -            regularizer_loss_out = tf.nn.l2_loss(self.W_out)
    -            regularizer_loss = self.lmbd*(regularizer_loss_fc1 + regularizer_loss_fc2 + regularizer_loss_out)
    -            
    -            self.loss = softmax_loss + regularizer_loss
     
    -    def create_accuracy(self):
    -        with tf.name_scope('accuracy'):
    -            probabilities = tf.nn.softmax(self.z_out)
    -            predictions = tf.argmax(probabilities, axis=1)
    -            labels = tf.argmax(self.Y, axis=1)
    -            
    -            correct_predictions = tf.equal(predictions, labels)
    -            correct_predictions = tf.cast(correct_predictions, tf.float32)
    -            self.accuracy = tf.reduce_mean(correct_predictions)
    -    
    -    def create_optimiser(self):
    -        with tf.name_scope('optimizer'):
    -            self.optimizer = tf.train.GradientDescentOptimizer(learning_rate=self.eta).minimize(self.loss, global_step=self.global_step)
    -            
    -    def weight_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.truncated_normal(shape, stddev=0.1)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -    
    -    def bias_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.constant(0.1, shape=shape)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -    
    -    def fit(self):
    -        data_indices = np.arange(self.n_inputs)
    +# ensure the same random numbers appear every time
    +np.random.seed(0)
     
    -        with tf.Session() as sess:
    -            sess.run(tf.global_variables_initializer())
    -            for i in range(self.epochs):
    -                for j in range(self.iterations):
    -                    chosen_datapoints = np.random.choice(data_indices, size=self.batch_size, replace=False)
    -                    batch_X, batch_Y = self.X_train[chosen_datapoints], self.Y_train[chosen_datapoints]
    -            
    -                    sess.run([DNN.loss, DNN.optimizer],
    -                        feed_dict={DNN.X: batch_X,
    -                                   DNN.Y: batch_Y})
    -                    accuracy = sess.run(DNN.accuracy,
    -                        feed_dict={DNN.X: batch_X,
    -                                   DNN.Y: batch_Y})
    -                    step = sess.run(DNN.global_step)
    +# display images in notebook
    +%matplotlib inline
    +plt.rcParams['figure.figsize'] = (12,12)
    +
    +
    +# download MNIST dataset
    +digits = datasets.load_digits()
    +
    +# define inputs and labels
    +inputs = digits.images
    +labels = digits.target
    +
    +print("inputs = (n_inputs, pixel_width, pixel_height) = " + str(inputs.shape))
    +print("labels = (n_inputs) = " + str(labels.shape))
    +
    +
    +# flatten the image
    +# the value -1 means dimension is inferred from the remaining dimensions: 8x8 = 64
    +n_inputs = len(inputs)
    +inputs = inputs.reshape(n_inputs, -1)
    +print("X = (n_inputs, n_features) = " + str(inputs.shape))
    +
    +
    +# choose some random images to display
    +indices = np.arange(n_inputs)
    +random_indices = np.random.choice(indices, size=5)
    +
    +for i, image in enumerate(digits.images[random_indices]):
    +    plt.subplot(1, 5, i+1)
    +    plt.axis('off')
    +    plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest')
    +    plt.title("Label: %d" % digits.target[random_indices[i]])
    +plt.show()
    +
    +

    + + +

    from tensorflow.keras.layers import Input
    +from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    +from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    +from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    +from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    +from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    +
    +from sklearn.model_selection import train_test_split
    +
    +# one-hot representation of labels
    +labels = to_categorical(labels)
    +
    +# split into train and test data
    +train_size = 0.8
    +test_size = 1 - train_size
    +X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,
    +                                                    test_size=test_size)
    +
    +

    + + +

    def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):
    +    model = Sequential()
    +    model.add(Dense(n_neurons_layer1, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
    +    model.add(Dense(n_neurons_layer2, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
    +    model.add(Dense(n_categories, activation='softmax'))
         
    -            self.train_loss, self.train_accuracy = sess.run([DNN.loss, DNN.accuracy],
    -                feed_dict={DNN.X: self.X_train,
    -                           DNN.Y: self.Y_train})
    +    sgd = optimizers.SGD(lr=eta)
    +    model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])
    +    
    +    return model
    +
    +

    + + +

    DNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
             
    -            self.test_loss, self.test_accuracy = sess.run([DNN.loss, DNN.accuracy],
    -                feed_dict={DNN.X: self.X_test,
    -                           DNN.Y: self.Y_test})
    +for i, eta in enumerate(eta_vals):
    +    for j, lmbd in enumerate(lmbd_vals):
    +        DNN = create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories,
    +                                         eta=eta, lmbd=lmbd)
    +        DNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0)
    +        scores = DNN.evaluate(X_test, Y_test)
    +        
    +        DNN_keras[i][j] = DNN
    +        
    +        print("Learning rate = ", eta)
    +        print("Lambda = ", lmbd)
    +        print("Test accuracy: %.3f" % scores[1])
    +        print()
    +
    +

    + + +

    # optional
    +# visual representation of grid search
    +# uses seaborn heatmap, could probably do this in matplotlib
    +import seaborn as sns
    +
    +sns.set()
    +
    +train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +
    +for i in range(len(eta_vals)):
    +    for j in range(len(lmbd_vals)):
    +        DNN = DNN_keras[i][j]
    +
    +        train_accuracy[i][j] = DNN.evaluate(X_train, Y_train)[1]
    +        test_accuracy[i][j] = DNN.evaluate(X_test, Y_test)[1]
    +
    +        
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Training Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
    +
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Test Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
     

    @@ -375,7 +417,7 @@ MathJax.Hub.Config({

  • 38
  • 39
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs030.html b/doc/pub/week41/html/._week41-bs030.html index 34a1449be..859f559e9 100644 --- a/doc/pub/week41/html/._week41-bs030.html +++ b/doc/pub/week41/html/._week41-bs030.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,80 +253,176 @@ MathJax.Hub.Config({ -

    Optimizing and using gradient descent

    +

    The Breast Cancer Data, now with Keras

    -

    epochs = 100
    -batch_size = 100
    -n_neurons_layer1 = 100
    -n_neurons_layer2 = 50
    -n_categories = 10
    -eta_vals = np.logspace(-5, 1, 7)
    -lmbd_vals = np.logspace(-5, 1, 7)
    -
    -

    - - -

    DNN_tf = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    -        
    -for i, eta in enumerate(eta_vals):
    -    for j, lmbd in enumerate(lmbd_vals):
    -        DNN = NeuralNetworkTensorflow(X_train, Y_train, X_test, Y_test,
    -                                      n_neurons_layer1, n_neurons_layer2, n_categories,
    -                                      epochs=epochs, batch_size=batch_size, eta=eta, lmbd=lmbd)
    -        DNN.fit()
    -        
    -        DNN_tf[i][j] = DNN
    -        
    -        print("Learning rate = ", eta)
    -        print("Lambda = ", lmbd)
    -        print("Test accuracy: %.3f" % DNN.test_accuracy)
    -        print()
    -
    -

    - - -

    # optional
    -# visual representation of grid search
    -# uses seaborn heatmap, could probably do this in matplotlib
    +
    import tensorflow as tf
    +from tensorflow.keras.layers import Input
    +from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    +from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    +from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    +from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    +from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    +import numpy as np
    +import matplotlib.pyplot as plt
     import seaborn as sns
    +from sklearn.model_selection import train_test_split as splitter
    +from sklearn.datasets import load_breast_cancer
    +import pickle
    +import os 
     
    -sns.set()
     
    -train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +"""Load breast cancer dataset"""
     
    -for i in range(len(eta_vals)):
    -    for j in range(len(lmbd_vals)):
    -        DNN = DNN_tf[i][j]
    +np.random.seed(0)        #create same seed for random number every time
     
    -        train_accuracy[i][j] = DNN.train_accuracy
    -        test_accuracy[i][j] = DNN.test_accuracy
    +cancer=load_breast_cancer()      #Download breast cancer dataset
     
    -        
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Training Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    +inputs=cancer.data                     #Feature matrix of 569 rows (samples) and 30 columns (parameters)
    +outputs=cancer.target                  #Label array of 569 rows (0 for benign and 1 for malignant)
    +labels=cancer.feature_names[0:30]
    +
    +print('The content of the breast cancer dataset is:')      #Print information about the datasets
    +print(labels)
    +print('-------------------------')
    +print("inputs =  " + str(inputs.shape))
    +print("outputs =  " + str(outputs.shape))
    +print("labels =  "+ str(labels.shape))
    +
    +x=inputs      #Reassign the Feature and Label matrices to other variables
    +y=outputs
    +
    +#%% 
    +
    +# Visualisation of dataset (for correlation analysis)
    +
    +plt.figure()
    +plt.scatter(x[:,0],x[:,2],s=40,c=y,cmap=plt.cm.Spectral)
    +plt.xlabel('Mean radius',fontweight='bold')
    +plt.ylabel('Mean perimeter',fontweight='bold')
     plt.show()
     
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Test Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    +plt.figure()
    +plt.scatter(x[:,5],x[:,6],s=40,c=y, cmap=plt.cm.Spectral)
    +plt.xlabel('Mean compactness',fontweight='bold')
    +plt.ylabel('Mean concavity',fontweight='bold')
     plt.show()
    -
    -

    - -

    # optional
    -# we can use log files to visualize our graph in Tensorboard
    -writer = tf.summary.FileWriter('logs/')
    -writer.add_graph(tf.get_default_graph())
    +
    +plt.figure()
    +plt.scatter(x[:,0],x[:,1],s=40,c=y,cmap=plt.cm.Spectral)
    +plt.xlabel('Mean radius',fontweight='bold')
    +plt.ylabel('Mean texture',fontweight='bold')
    +plt.show()
    +
    +plt.figure()
    +plt.scatter(x[:,2],x[:,1],s=40,c=y,cmap=plt.cm.Spectral)
    +plt.xlabel('Mean perimeter',fontweight='bold')
    +plt.ylabel('Mean compactness',fontweight='bold')
    +plt.show()
    +
    +
    +# Generate training and testing datasets
    +
    +#Select features relevant to classification (texture,perimeter,compactness and symmetery) 
    +#and add to input matrix
    +
    +temp1=np.reshape(x[:,1],(len(x[:,1]),1))
    +temp2=np.reshape(x[:,2],(len(x[:,2]),1))
    +X=np.hstack((temp1,temp2))      
    +temp=np.reshape(x[:,5],(len(x[:,5]),1))
    +X=np.hstack((X,temp))       
    +temp=np.reshape(x[:,8],(len(x[:,8]),1))
    +X=np.hstack((X,temp))       
    +
    +X_train,X_test,y_train,y_test=splitter(X,y,test_size=0.1)   #Split datasets into training and testing
    +
    +y_train=to_categorical(y_train)     #Convert labels to categorical when using categorical cross entropy
    +y_test=to_categorical(y_test)
    +
    +del temp1,temp2,temp
    +
    +# %%
    +
    +# Define tunable parameters"
    +
    +eta=np.logspace(-3,-1,3)                    #Define vector of learning rates (parameter to SGD optimiser)
    +lamda=0.01                                  #Define hyperparameter
    +n_layers=2                                  #Define number of hidden layers in the model
    +n_neuron=np.logspace(0,3,4,dtype=int)       #Define number of neurons per layer
    +epochs=100                                   #Number of reiterations over the input data
    +batch_size=100                              #Number of samples per gradient update
    +
    +# %%
    +
    +"""Define function to return Deep Neural Network model"""
    +
    +def NN_model(inputsize,n_layers,n_neuron,eta,lamda):
    +    model=Sequential()      
    +    for i in range(n_layers):       #Run loop to add hidden layers to the model
    +        if (i==0):                  #First layer requires input dimensions
    +            model.add(Dense(n_neuron,activation='relu',kernel_regularizer=regularizers.l2(lamda),input_dim=inputsize))
    +        else:                       #Subsequent layers are capable of automatic shape inferencing
    +            model.add(Dense(n_neuron,activation='relu',kernel_regularizer=regularizers.l2(lamda)))
    +    model.add(Dense(2,activation='softmax'))  #2 outputs - ordered and disordered (softmax for prob)
    +    sgd=optimizers.SGD(lr=eta)
    +    model.compile(loss='categorical_crossentropy',optimizer=sgd,metrics=['accuracy'])
    +    return model
    +
    +    
    +Train_accuracy=np.zeros((len(n_neuron),len(eta)))      #Define matrices to store accuracy scores as a function
    +Test_accuracy=np.zeros((len(n_neuron),len(eta)))       #of learning rate and number of hidden neurons for 
    +
    +for i in range(len(n_neuron)):     #run loops over hidden neurons and learning rates to calculate 
    +    for j in range(len(eta)):      #accuracy scores 
    +        DNN_model=NN_model(X_train.shape[1],n_layers,n_neuron[i],eta[j],lamda)
    +        DNN_model.fit(X_train,y_train,epochs=epochs,batch_size=batch_size,verbose=1)
    +        Train_accuracy[i,j]=DNN_model.evaluate(X_train,y_train)[1]
    +        Test_accuracy[i,j]=DNN_model.evaluate(X_test,y_test)[1]
    +               
    +
    +def plot_data(x,y,data,title=None):
    +
    +    # plot results
    +    fontsize=16
    +
    +
    +    fig = plt.figure()
    +    ax = fig.add_subplot(111)
    +    cax = ax.matshow(data, interpolation='nearest', vmin=0, vmax=1)
    +    
    +    cbar=fig.colorbar(cax)
    +    cbar.ax.set_ylabel('accuracy (%)',rotation=90,fontsize=fontsize)
    +    cbar.set_ticks([0,.2,.4,0.6,0.8,1.0])
    +    cbar.set_ticklabels(['0%','20%','40%','60%','80%','100%'])
    +
    +    # put text on matrix elements
    +    for i, x_val in enumerate(np.arange(len(x))):
    +        for j, y_val in enumerate(np.arange(len(y))):
    +            c = "${0:.1f}\\%$".format( 100*data[j,i])  
    +            ax.text(x_val, y_val, c, va='center', ha='center')
    +
    +    # convert axis vaues to to string labels
    +    x=[str(i) for i in x]
    +    y=[str(i) for i in y]
    +
    +
    +    ax.set_xticklabels(['']+x)
    +    ax.set_yticklabels(['']+y)
    +
    +    ax.set_xlabel('$\\mathrm{learning\\ rate}$',fontsize=fontsize)
    +    ax.set_ylabel('$\\mathrm{hidden\\ neurons}$',fontsize=fontsize)
    +    if title is not None:
    +        ax.set_title(title)
    +
    +    plt.tight_layout()
    +
    +    plt.show()
    +    
    +plot_data(eta,n_neuron,Train_accuracy, 'training')
    +plot_data(eta,n_neuron,Test_accuracy, 'testing')
     

    @@ -309,7 +450,7 @@ writer.add_graph(tf39

  • 40
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs031.html b/doc/pub/week41/html/._week41-bs031.html index a7841dc59..a4fb5408d 100644 --- a/doc/pub/week41/html/._week41-bs031.html +++ b/doc/pub/week41/html/._week41-bs031.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,106 +253,30 @@ MathJax.Hub.Config({ -

    Using Keras

    +

    Fine-tuning neural network hyperparameters

    -Keras is a high level neural network -that supports Tensorflow, CTNK and Theano as backends. -If you have Tensorflow installed Keras is available through the tf.keras module. -If you have Anaconda installed you may run the following command -

    +The flexibility of neural networks is also one of their main +drawbacks: there are many hyperparameters to tweak. Not only can you +use any imaginable network topology (how neurons/nodes are interconnected), +but even in a simple FFNN you can change the number of layers, the +number of neurons per layer, the type of activation function to use in +each layer, the weight initialization logic, the stochastic gradient optmized and much more. How do you +know what combination of hyperparameters is the best for your task? - -

    conda install keras
    -
    -

    -Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager: +

      +
    • You can use grid search with cross-validation to find the right hyperparameters.
    • +
    -

    +However,since there are many hyperparameters to tune, and since +training a neural network on a large dataset takes a lot of time, you +will only be able to explore a tiny part of the hyperparameter space. - -

    pip install keras
    -
    -

    -or look up the instructions here. +

      +
    • You can use randomized search.
    • +
    • Or use tools like Oscar, which implements more complex algorithms to help you find a good set of hyperparameters quickly.
    • +
    -

    - - -

    import tensorflow as tf
    -from tensorflow.keras.layers import Input
    -from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    -from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    -from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    -from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    -from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    -
    -def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):
    -    model = Sequential()
    -    model.add(Dense(n_neurons_layer1, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
    -    model.add(Dense(n_neurons_layer2, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
    -    model.add(Dense(n_categories, activation='softmax'))
    -    
    -    sgd = optimizers.SGD(lr=eta)
    -    model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])
    -    
    -    return model
    -
    -

    - - -

    DNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    -        
    -for i, eta in enumerate(eta_vals):
    -    for j, lmbd in enumerate(lmbd_vals):
    -        DNN = create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories,
    -                                         eta=eta, lmbd=lmbd)
    -        DNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0)
    -        scores = DNN.evaluate(X_test, Y_test)
    -        
    -        DNN_keras[i][j] = DNN
    -        
    -        print("Learning rate = ", eta)
    -        print("Lambda = ", lmbd)
    -        print("Test accuracy: %.3f" % scores[1])
    -        print()
    -
    -

    - - -

    # optional
    -# visual representation of grid search
    -# uses seaborn heatmap, could probably do this in matplotlib
    -import seaborn as sns
    -
    -sns.set()
    -
    -train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -
    -for i in range(len(eta_vals)):
    -    for j in range(len(lmbd_vals)):
    -        DNN = DNN_keras[i][j]
    -
    -        train_accuracy[i][j] = DNN.evaluate(X_train, Y_train)[1]
    -        test_accuracy[i][j] = DNN.evaluate(X_test, Y_test)[1]
    -
    -        
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Training Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    -
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Test Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    -
    -

    diff --git a/doc/pub/week41/html/._week41-bs032.html b/doc/pub/week41/html/._week41-bs032.html index c389e464b..a7c918bba 100644 --- a/doc/pub/week41/html/._week41-bs032.html +++ b/doc/pub/week41/html/._week41-bs032.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,177 +253,23 @@ MathJax.Hub.Config({ -

    The Breast Cancer Data, now with Keras

    +

    Hidden layers

    +For many problems you can start with just one or two hidden layers and it will work just fine. +For the MNIST data set you ca easily get a high accuracy using just one hidden layer with a +few hundred neurons. +You can reach for this data set above 98% accuracy using two hidden layers with the same total amount of +neurons, in roughly the same amount of training time. - -

    import tensorflow as tf
    -from tensorflow.keras.layers import Input
    -from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    -from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    -from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    -from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    -from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    -import numpy as np
    -import matplotlib.pyplot as plt
    -import seaborn as sns
    -from sklearn.model_selection import train_test_split as splitter
    -from sklearn.datasets import load_breast_cancer
    -import pickle
    -import os 
    +

    +For more complex problems, you can gradually +ramp up the number of hidden layers, until you start overfitting the training set. Very complex tasks, such +as large image classification or speech recognition, typically require networks with dozens of layers +and they need a huge amount +of training data. However, you will rarely have to train such networks from scratch: it is much more +common to reuse parts of a pretrained state-of-the-art network that performs a similar task. - -"""Load breast cancer dataset""" - -np.random.seed(0) #create same seed for random number every time - -cancer=load_breast_cancer() #Download breast cancer dataset - -inputs=cancer.data #Feature matrix of 569 rows (samples) and 30 columns (parameters) -outputs=cancer.target #Label array of 569 rows (0 for benign and 1 for malignant) -labels=cancer.feature_names[0:30] - -print('The content of the breast cancer dataset is:') #Print information about the datasets -print(labels) -print('-------------------------') -print("inputs = " + str(inputs.shape)) -print("outputs = " + str(outputs.shape)) -print("labels = "+ str(labels.shape)) - -x=inputs #Reassign the Feature and Label matrices to other variables -y=outputs - -#%% - -# Visualisation of dataset (for correlation analysis) - -plt.figure() -plt.scatter(x[:,0],x[:,2],s=40,c=y,cmap=plt.cm.Spectral) -plt.xlabel('Mean radius',fontweight='bold') -plt.ylabel('Mean perimeter',fontweight='bold') -plt.show() - -plt.figure() -plt.scatter(x[:,5],x[:,6],s=40,c=y, cmap=plt.cm.Spectral) -plt.xlabel('Mean compactness',fontweight='bold') -plt.ylabel('Mean concavity',fontweight='bold') -plt.show() - - -plt.figure() -plt.scatter(x[:,0],x[:,1],s=40,c=y,cmap=plt.cm.Spectral) -plt.xlabel('Mean radius',fontweight='bold') -plt.ylabel('Mean texture',fontweight='bold') -plt.show() - -plt.figure() -plt.scatter(x[:,2],x[:,1],s=40,c=y,cmap=plt.cm.Spectral) -plt.xlabel('Mean perimeter',fontweight='bold') -plt.ylabel('Mean compactness',fontweight='bold') -plt.show() - - -# Generate training and testing datasets - -#Select features relevant to classification (texture,perimeter,compactness and symmetery) -#and add to input matrix - -temp1=np.reshape(x[:,1],(len(x[:,1]),1)) -temp2=np.reshape(x[:,2],(len(x[:,2]),1)) -X=np.hstack((temp1,temp2)) -temp=np.reshape(x[:,5],(len(x[:,5]),1)) -X=np.hstack((X,temp)) -temp=np.reshape(x[:,8],(len(x[:,8]),1)) -X=np.hstack((X,temp)) - -X_train,X_test,y_train,y_test=splitter(X,y,test_size=0.1) #Split datasets into training and testing - -y_train=to_categorical(y_train) #Convert labels to categorical when using categorical cross entropy -y_test=to_categorical(y_test) - -del temp1,temp2,temp - -# %% - -# Define tunable parameters" - -eta=np.logspace(-3,-1,3) #Define vector of learning rates (parameter to SGD optimiser) -lamda=0.01 #Define hyperparameter -n_layers=2 #Define number of hidden layers in the model -n_neuron=np.logspace(0,3,4,dtype=int) #Define number of neurons per layer -epochs=100 #Number of reiterations over the input data -batch_size=100 #Number of samples per gradient update - -# %% - -"""Define function to return Deep Neural Network model""" - -def NN_model(inputsize,n_layers,n_neuron,eta,lamda): - model=Sequential() - for i in range(n_layers): #Run loop to add hidden layers to the model - if (i==0): #First layer requires input dimensions - model.add(Dense(n_neuron,activation='relu',kernel_regularizer=regularizers.l2(lamda),input_dim=inputsize)) - else: #Subsequent layers are capable of automatic shape inferencing - model.add(Dense(n_neuron,activation='relu',kernel_regularizer=regularizers.l2(lamda))) - model.add(Dense(2,activation='softmax')) #2 outputs - ordered and disordered (softmax for prob) - sgd=optimizers.SGD(lr=eta) - model.compile(loss='categorical_crossentropy',optimizer=sgd,metrics=['accuracy']) - return model - - -Train_accuracy=np.zeros((len(n_neuron),len(eta))) #Define matrices to store accuracy scores as a function -Test_accuracy=np.zeros((len(n_neuron),len(eta))) #of learning rate and number of hidden neurons for - -for i in range(len(n_neuron)): #run loops over hidden neurons and learning rates to calculate - for j in range(len(eta)): #accuracy scores - DNN_model=NN_model(X_train.shape[1],n_layers,n_neuron[i],eta[j],lamda) - DNN_model.fit(X_train,y_train,epochs=epochs,batch_size=batch_size,verbose=1) - Train_accuracy[i,j]=DNN_model.evaluate(X_train,y_train)[1] - Test_accuracy[i,j]=DNN_model.evaluate(X_test,y_test)[1] - - -def plot_data(x,y,data,title=None): - - # plot results - fontsize=16 - - - fig = plt.figure() - ax = fig.add_subplot(111) - cax = ax.matshow(data, interpolation='nearest', vmin=0, vmax=1) - - cbar=fig.colorbar(cax) - cbar.ax.set_ylabel('accuracy (%)',rotation=90,fontsize=fontsize) - cbar.set_ticks([0,.2,.4,0.6,0.8,1.0]) - cbar.set_ticklabels(['0%','20%','40%','60%','80%','100%']) - - # put text on matrix elements - for i, x_val in enumerate(np.arange(len(x))): - for j, y_val in enumerate(np.arange(len(y))): - c = "${0:.1f}\\%$".format( 100*data[j,i]) - ax.text(x_val, y_val, c, va='center', ha='center') - - # convert axis vaues to to string labels - x=[str(i) for i in x] - y=[str(i) for i in y] - - - ax.set_xticklabels(['']+x) - ax.set_yticklabels(['']+y) - - ax.set_xlabel('$\\mathrm{learning\\ rate}$',fontsize=fontsize) - ax.set_ylabel('$\\mathrm{hidden\\ neurons}$',fontsize=fontsize) - if title is not None: - ax.set_title(title) - - plt.tight_layout() - - plt.show() - -plot_data(eta,n_neuron,Train_accuracy, 'training') -plot_data(eta,n_neuron,Test_accuracy, 'testing') -

    @@ -405,7 +296,7 @@ plot_data(eta,n_neuron,Test_accuracy, 'testing&

  • 41
  • 42
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs033.html b/doc/pub/week41/html/._week41-bs033.html index f1eccedc3..c4191231b 100644 --- a/doc/pub/week41/html/._week41-bs033.html +++ b/doc/pub/week41/html/._week41-bs033.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -260,7 +305,7 @@ learn at widely different speeds
  • 42
  • 43
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs034.html b/doc/pub/week41/html/._week41-bs034.html index 62f905f8b..5c7bbfdf3 100644 --- a/doc/pub/week41/html/._week41-bs034.html +++ b/doc/pub/week41/html/._week41-bs034.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -206,36 +251,24 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Is the Logistic activation function (Sigmoid) our choice?

    +

    More on activation functions, output layers

    -Although this unfortunate behavior has been empirically observed for -quite a while (it was one of the reasons why deep neural networks were -mostly abandoned for a long time), it is only around 2010 that -significant progress was made in understanding it. +In most cases you can use the ReLU activation function in the hidden layers (or one of its variants).

    -A paper titled Understanding the Difficulty of Training Deep -Feedforward Neural Networks by Xavier Glorot and Yoshua Bengio found that -the problems with the popular logistic -sigmoid activation function and the weight initialization technique -that was most popular at the time, namely random initialization using -a normal distribution with a mean of 0 and a standard deviation of -1. +It is a bit faster to compute than other activation functions, and the gradient descent optimization does in general not get stuck.

    -They showed that with this activation function and this -initialization scheme, the variance of the outputs of each layer is -much greater than the variance of its inputs. Going forward in the -network, the variance keeps increasing after each layer until the -activation function saturates at the top layers. This is actually made -worse by the fact that the logistic function has a mean of 0.5, not 0 -(the hyperbolic tangent function has a mean of 0 and behaves slightly -better than the logistic function in deep networks). +For the output layer: + +

      +
    • For classification the softmax activation function is generally a good choice for classification tasks (when the classes are mutually exclusive).
    • +
    • For regression tasks, you can simply use no activation function at all.
    • +
    -

      @@ -261,7 +294,7 @@ better than the logistic function in deep networks).
    • 43
    • 44
    • ...
    • -
    • 46
    • +
    • 62
    • »
    diff --git a/doc/pub/week41/html/._week41-bs035.html b/doc/pub/week41/html/._week41-bs035.html index 1b36d8f28..9adf3e0d5 100644 --- a/doc/pub/week41/html/._week41-bs035.html +++ b/doc/pub/week41/html/._week41-bs035.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -206,40 +251,34 @@ MathJax.Hub.Config({

     

     

     

    - + -

    The derivative of the Logistic funtion

    +

    Is the Logistic activation function (Sigmoid) our choice?

    -Looking at the logistic activation function, when inputs become large -(negative or positive), the function saturates at 0 or 1, with a -derivative extremely close to 0. Thus when backpropagation kicks in, -it has virtually no gradient to propagate back through the network, -and what little gradient exists keeps getting diluted as -backpropagation progresses down through the top layers, so there is -really nothing left for the lower layers. +Although this unfortunate behavior has been empirically observed for +quite a while (it was one of the reasons why deep neural networks were +mostly abandoned for a long time), it is only around 2010 that +significant progress was made in understanding it.

    -In their paper, Glorot and Bengio propose a way to significantly -alleviate this problem. We need the signal to flow properly in both -directions: in the forward direction when making predictions, and in -the reverse direction when backpropagating gradients. We don’t want -the signal to die out, nor do we want it to explode and saturate. For -the signal to flow properly, the authors argue that we need the -variance of the outputs of each layer to be equal to the variance of -its inputs, and we also need the gradients to have equal variance -before and after flowing through a layer in the reverse direction. +A paper titled Understanding the Difficulty of Training Deep +Feedforward Neural Networks by Xavier Glorot and Yoshua Bengio found that +the problems with the popular logistic +sigmoid activation function and the weight initialization technique +that was most popular at the time, namely random initialization using +a normal distribution with a mean of 0 and a standard deviation of +1.

    -One of the insights in the 2010 paper by Glorot and Bengio was that -the vanishing/exploding gradients problems were in part due to a poor -choice of activation function. Until then most people had assumed that -if Nature had chosen to use roughly sigmoid activation functions in -biological neurons, they must be an excellent choice. But it turns out -that other activation functions behave much better in deep neural -networks, in particular the ReLU activation function, mostly because -it does not saturate for positive values (and also because it is quite -fast to compute). +They showed that with this activation function and this +initialization scheme, the variance of the outputs of each layer is +much greater than the variance of its inputs. Going forward in the +network, the variance keeps increasing after each layer until the +activation function saturates at the top layers. This is actually made +worse by the fact that the logistic function has a mean of 0.5, not 0 +(the hyperbolic tangent function has a mean of 0 and behaves slightly +better than the logistic function in deep networks).

    @@ -267,7 +306,7 @@ fast to compute).

  • 44
  • 45
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs036.html b/doc/pub/week41/html/._week41-bs036.html index b0b85507f..7380c044e 100644 --- a/doc/pub/week41/html/._week41-bs036.html +++ b/doc/pub/week41/html/._week41-bs036.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,29 +253,38 @@ MathJax.Hub.Config({ -

    The RELU function family

    +

    The derivative of the Logistic funtion

    -The ReLU activation function suffers from a problem known as the dying -ReLUs: during training, some neurons effectively die, meaning they -stop outputting anything other than 0. +Looking at the logistic activation function, when inputs become large +(negative or positive), the function saturates at 0 or 1, with a +derivative extremely close to 0. Thus when backpropagation kicks in, +it has virtually no gradient to propagate back through the network, +and what little gradient exists keeps getting diluted as +backpropagation progresses down through the top layers, so there is +really nothing left for the lower layers.

    -In some cases, you may find that half of your network’s neurons are -dead, especially if you used a large learning rate. During training, -if a neuron’s weights get updated such that the weighted sum of the -neuron’s inputs is negative, it will start outputting 0. When this -happen, the neuron is unlikely to come back to life since the gradient -of the ReLU function is 0 when its input is negative. +In their paper, Glorot and Bengio propose a way to significantly +alleviate this problem. We need the signal to flow properly in both +directions: in the forward direction when making predictions, and in +the reverse direction when backpropagating gradients. We don’t want +the signal to die out, nor do we want it to explode and saturate. For +the signal to flow properly, the authors argue that we need the +variance of the outputs of each layer to be equal to the variance of +its inputs, and we also need the gradients to have equal variance +before and after flowing through a layer in the reverse direction.

    -To solve this problem, nowadays practitioners use a variant of the ReLU -function, such as the leaky ReLU discussed above or the so-called -exponential linear unit (ELU) function - -$$ -ELU(z) = \left\{\begin{array}{cc} \alpha\left( \exp{(z)}-1\right) & z < 0,\\ z & z \ge 0.\end{array}\right. -$$ +One of the insights in the 2010 paper by Glorot and Bengio was that +the vanishing/exploding gradients problems were in part due to a poor +choice of activation function. Until then most people had assumed that +if Nature had chosen to use roughly sigmoid activation functions in +biological neurons, they must be an excellent choice. But it turns out +that other activation functions behave much better in deep neural +networks, in particular the ReLU activation function, mostly because +it does not saturate for positive values (and also because it is quite +fast to compute).

    @@ -257,6 +311,8 @@ $$

  • 44
  • 45
  • 46
  • +
  • ...
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs037.html b/doc/pub/week41/html/._week41-bs037.html index e7f838e61..756a443c6 100644 --- a/doc/pub/week41/html/._week41-bs037.html +++ b/doc/pub/week41/html/._week41-bs037.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,22 +253,29 @@ MathJax.Hub.Config({ -

    Which activation function should we use?

    +

    The RELU function family

    -In general it seems that the ELU activation function is better than -the leaky ReLU function (and its variants), which is better than -ReLU. ReLU performs better than \( \tanh \) which in turn performs better -than the logistic function. +The ReLU activation function suffers from a problem known as the dying +ReLUs: during training, some neurons effectively die, meaning they +stop outputting anything other than 0.

    -If runtime -performance is an issue, then you may opt for the leaky ReLU function over the -ELU function If you don’t -want to tweak yet another hyperparameter, you may just use the default -\( \alpha \) of \( 0.01 \) for the leaky ReLU, and \( 1 \) for ELU. If you have -spare time and computing power, you can use cross-validation or -bootstrap to evaluate other activation functions. +In some cases, you may find that half of your network’s neurons are +dead, especially if you used a large learning rate. During training, +if a neuron’s weights get updated such that the weighted sum of the +neuron’s inputs is negative, it will start outputting 0. When this +happen, the neuron is unlikely to come back to life since the gradient +of the ReLU function is 0 when its input is negative. + +

    +To solve this problem, nowadays practitioners use a variant of the ReLU +function, such as the leaky ReLU discussed above or the so-called +exponential linear unit (ELU) function + +$$ +ELU(z) = \left\{\begin{array}{cc} \alpha\left( \exp{(z)}-1\right) & z < 0,\\ z & z \ge 0.\end{array}\right. +$$

    @@ -249,6 +301,9 @@ bootstrap to evaluate other activation functions.

  • 44
  • 45
  • 46
  • +
  • 47
  • +
  • ...
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs038.html b/doc/pub/week41/html/._week41-bs038.html index c6047b465..3ae958001 100644 --- a/doc/pub/week41/html/._week41-bs038.html +++ b/doc/pub/week41/html/._week41-bs038.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -206,46 +251,24 @@ MathJax.Hub.Config({

     

     

     

    - + -

    A top-down perspective on Neural networks

    +

    Which activation function should we use?

    -The first thing we would like to do is divide the data into two or three -parts. A training set, a validation or dev (development) set, and a -test set. The test set is the data on which we want to make -predictions. The dev set is a subset of the training data we use to -check how well we are doing out-of-sample, after training the model on -the training dataset. We use the validation error as a proxy for the -test error in order to make tweaks to our model. It is crucial that we -do not use any of the test data to train the algorithm. This is a -cardinal sin in ML. Then: - -

      -
    • Estimate optimal error rate
    • -
    • Minimize underfitting (bias) on training data set.
    • -
    • Make sure you are not overfitting.
    • -
    - -If the validation and test sets are drawn from the same distributions, -then a good performance on the validation set should lead to similarly -good performance on the test set. +In general it seems that the ELU activation function is better than +the leaky ReLU function (and its variants), which is better than +ReLU. ReLU performs better than \( \tanh \) which in turn performs better +than the logistic function.

    -However, sometimes -the training data and test data differ in subtle ways because, for -example, they are collected using slightly different methods, or -because it is cheaper to collect data in one way versus another. In -this case, there can be a mismatch between the training and test -data. This can lead to the neural network overfitting these small -differences between the test and training sets, and a poor performance -on the test set despite having a good performance on the validation -set. To rectify this, Andrew Ng suggests making two validation or dev -sets, one constructed from the training data and one constructed from -the test data. The difference between the performance of the algorithm -on these two validation sets quantifies the train-test mismatch. This -can serve as another important diagnostic when using DNNs for -supervised learning. +If runtime +performance is an issue, then you may opt for the leaky ReLU function over the +ELU function If you don’t +want to tweak yet another hyperparameter, you may just use the default +\( \alpha \) of \( 0.01 \) for the leaky ReLU, and \( 1 \) for ELU. If you have +spare time and computing power, you can use cross-validation or +bootstrap to evaluate other activation functions.

    @@ -270,6 +293,10 @@ supervised learning.

  • 44
  • 45
  • 46
  • +
  • 47
  • +
  • 48
  • +
  • ...
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs039.html b/doc/pub/week41/html/._week41-bs039.html index ca8b4f3d2..63ea1166f 100644 --- a/doc/pub/week41/html/._week41-bs039.html +++ b/doc/pub/week41/html/._week41-bs039.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,28 +253,21 @@ MathJax.Hub.Config({ -

    Limitations of supervised learning with deep networks

    +

    Batch Normalization

    -Like all statistical methods, supervised learning using neural -networks has important limitations. This is especially important when -one seeks to apply these methods, especially to physics problems. Like -all tools, DNNs are not a universal solution. Often, the same or -better performance on a task can be achieved by using a few -hand-engineered features (or even a collection of random -features). +Batch Normalization +aims to address the vanishing/exploding gradients problems, and more generally the problem that the +distribution of each layer’s inputs changes during training, as the parameters of the previous layers change.

    -Here we list some of the important limitations of supervised neural network based models. - -

      -
    • Need labeled data. All supervised learning methods, DNNs for supervised learning require labeled data. Often, labeled data is harder to acquire than unlabeled data (e.g. one must pay for human experts to label images).
    • -
    • Supervised neural networks are extremely data intensive. DNNs are data hungry. They perform best when data is plentiful. This is doubly so for supervised methods where the data must also be labeled. The utility of DNNs is extremely limited if data is hard to acquire or the datasets are small (hundreds to a few thousand samples). In this case, the performance of other methods that utilize hand-engineered features can exceed that of DNNs.
    • -
    • Homogeneous data. Almost all DNNs deal with homogeneous data of one type. It is very hard to design architectures that mix and match data types (i.e. some continuous variables, some discrete variables, some time series). In applications beyond images, video, and language, this is often what is required. In contrast, ensemble models like random forests or gradient-boosted trees have no difficulty handling mixed data types.
    • -
    • Many problems are not about prediction. In natural science we are often interested in learning something about the underlying distribution that generates the data. In this case, it is often difficult to cast these ideas in a supervised learning setting. While the problems are related, it is possible to make good predictions with a wrong model. The model might or might not be useful for understanding the underlying science.
    • -
    - -Some of these remarks are particular to DNNs, others are shared by all supervised learning methods. This motivates the use of unsupervised methods which in part circumvent these problems. +The technique consists of adding an operation in the model just before the activation function of each +layer, simply zero-centering and normalizing the inputs, then scaling and shifting the result using two new +parameters per layer (one for scaling, the other for shifting). In other words, this operation lets the model +learn the optimal scale and mean of the inputs for each layer. +In order to zero-center and normalize the inputs, the algorithm needs to estimate the inputs’ mean and +standard deviation. It does so by evaluating the mean and standard deviation of the inputs over the current +mini-batch, from this the name batch normalization.

    @@ -253,6 +291,11 @@ Some of these remarks are particular to DNNs, others are shared by all supervise

  • 44
  • 45
  • 46
  • +
  • 47
  • +
  • 48
  • +
  • 49
  • +
  • ...
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs040.html b/doc/pub/week41/html/._week41-bs040.html index 150f15632..4225da0e3 100644 --- a/doc/pub/week41/html/._week41-bs040.html +++ b/doc/pub/week41/html/._week41-bs040.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,41 +253,17 @@ MathJax.Hub.Config({ -

    Convolutional Neural Networks (recognizing images)

    +

    Dropout

    -Convolutional neural networks (CNNs) were developed during the last -decade of the previous century, with a focus on character recognition -tasks. Nowadays, CNNs are a central element in the spectacular success -of deep learning methods. The success in for example image -classifications have made them a central tool for most machine -learning practitioners. +It is a fairly simple algorithm: at every training step, every neuron (including the input neurons but +excluding the output neurons) has a probability \( p \) of being temporarily dropped out, meaning it will be +entirely ignored during this training step, but it may be active during the next step.

    -CNNs are very similar to ordinary Neural Networks. -They are made up of neurons that have learnable weights and -biases. Each neuron receives some inputs, performs a dot product and -optionally follows it with a non-linearity. The whole network still -expresses a single differentiable score function: from the raw image -pixels on one end to class scores at the other. And they still have a -loss function (for example Softmax) on the last (fully-connected) layer -and all the tips/tricks we developed for learning regular Neural -Networks still apply (back propagation, gradient descent etc etc). - -

    -What is the difference? CNN architectures make the explicit assumption that -the inputs are images, which allows us to encode certain properties -into the architecture. These then make the forward function more -efficient to implement and vastly reduce the amount of parameters in -the network. - -

    -Here we provide only a superficial overview, for the more interested, we recommend highly the course -IN5400 – Machine Learning for Image Analysis -and the slides of CS231. - -

    -Another good read is the article here https://arxiv.org/pdf/1603.07285.pdf. +The +hyperparameter \( p \) is called the dropout rate, and it is typically set to 50%. After training, the neurons are not dropped anymore. + It is viewed as one of the most popular regularization techniques.

    @@ -265,6 +286,12 @@ Another good read is the article here 44

  • 45
  • 46
  • +
  • 47
  • +
  • 48
  • +
  • 49
  • +
  • 50
  • +
  • ...
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs041.html b/doc/pub/week41/html/._week41-bs041.html index a0c67248e..f3932ba09 100644 --- a/doc/pub/week41/html/._week41-bs041.html +++ b/doc/pub/week41/html/._week41-bs041.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,30 +253,19 @@ MathJax.Hub.Config({ -

    Regular NNs don’t scale well to full images

    +

    Gradient Clipping

    -As an example, consider -an image of size \( 32\times 32\times 3 \) (32 wide, 32 high, 3 color channels), so a -single fully-connected neuron in a first hidden layer of a regular -Neural Network would have \( 32\times 32\times 3 = 3072 \) weights. This amount still -seems manageable, but clearly this fully-connected structure does not -scale to larger images. For example, an image of more respectable -size, say \( 200\times 200\times 3 \), would lead to neurons that have -\( 200\times 200\times 3 = 120,000 \) weights. +A popular technique to lessen the exploding gradients problem is to simply clip the gradients during +backpropagation so that they never exceed some threshold (this is mostly useful for recurrent neural +networks).

    -We could have -several such neurons, and the parameters would add up quickly! Clearly, -this full connectivity is wasteful and the huge number of parameters -would quickly lead to possible overfitting. +This technique is called Gradient Clipping.

    -

    -
    -

    Figure 1: A regular 3-layer Neural Network.

    -

    -
    +In general however, Batch +Normalization is preferred.

    @@ -253,6 +287,13 @@ would quickly lead to possible overfitting.

  • 44
  • 45
  • 46
  • +
  • 47
  • +
  • 48
  • +
  • 49
  • +
  • 50
  • +
  • 51
  • +
  • ...
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs042.html b/doc/pub/week41/html/._week41-bs042.html index b42005ca2..eb4a91bfb 100644 --- a/doc/pub/week41/html/._week41-bs042.html +++ b/doc/pub/week41/html/._week41-bs042.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -206,44 +251,46 @@ MathJax.Hub.Config({

     

     

     

    - + -

    3D volumes of neurons

    +

    A top-down perspective on Neural networks

    -Convolutional Neural Networks take advantage of the fact that the -input consists of images and they constrain the architecture in a more -sensible way. +The first thing we would like to do is divide the data into two or three +parts. A training set, a validation or dev (development) set, and a +test set. The test set is the data on which we want to make +predictions. The dev set is a subset of the training data we use to +check how well we are doing out-of-sample, after training the model on +the training dataset. We use the validation error as a proxy for the +test error in order to make tweaks to our model. It is crucial that we +do not use any of the test data to train the algorithm. This is a +cardinal sin in ML. Then: + +

      +
    • Estimate optimal error rate
    • +
    • Minimize underfitting (bias) on training data set.
    • +
    • Make sure you are not overfitting.
    • +
    + +If the validation and test sets are drawn from the same distributions, +then a good performance on the validation set should lead to similarly +good performance on the test set.

    -In particular, unlike a regular Neural Network, the -layers of a CNN have neurons arranged in 3 dimensions: width, -height, depth. (Note that the word depth here refers to the third -dimension of an activation volume, not to the depth of a full Neural -Network, which can refer to the total number of layers in a network.) - -

    -To understand it better, the above example of an image -with an input volume of -activations has dimensions \( 32\times 32\times 3 \) (width, height, -depth respectively). - -

    -The neurons in a layer will -only be connected to a small region of the layer before it, instead of -all of the neurons in a fully-connected manner. Moreover, the final -output layer could for this specific image have dimensions \( 1\times 1 \times 10 \), -because by the -end of the CNN architecture we will reduce the full image into a -single vector of class scores, arranged along the depth -dimension. - -

    -

    -
    -

    Figure 2: A CNN arranges its neurons in three dimensions (width, height, depth), as visualized in one of the layers. Every layer of a CNN transforms the 3D input volume to a 3D output volume of neuron activations. In this example, the red input layer holds the image, so its width and height would be the dimensions of the image, and the depth would be 3 (Red, Green, Blue channels).

    -

    -
    +However, sometimes +the training data and test data differ in subtle ways because, for +example, they are collected using slightly different methods, or +because it is cheaper to collect data in one way versus another. In +this case, there can be a mismatch between the training and test +data. This can lead to the neural network overfitting these small +differences between the test and training sets, and a poor performance +on the test set despite having a good performance on the validation +set. To rectify this, Andrew Ng suggests making two validation or dev +sets, one constructed from the training data and one constructed from +the test data. The difference between the performance of the algorithm +on these two validation sets quantifies the train-test mismatch. This +can serve as another important diagnostic when using DNNs for +supervised learning.

    @@ -264,6 +311,14 @@ dimension.

  • 44
  • 45
  • 46
  • +
  • 47
  • +
  • 48
  • +
  • 49
  • +
  • 50
  • +
  • 51
  • +
  • 52
  • +
  • ...
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs043.html b/doc/pub/week41/html/._week41-bs043.html index 5d103dcab..9c6a267d9 100644 --- a/doc/pub/week41/html/._week41-bs043.html +++ b/doc/pub/week41/html/._week41-bs043.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -206,29 +251,32 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Layers used to build CNNs

    +

    Limitations of supervised learning with deep networks

    -A simple CNN is a sequence of layers, and every layer of a CNN -transforms one volume of activations to another through a -differentiable function. We use three main types of layers to build -CNN architectures: Convolutional Layer, Pooling Layer, and -Fully-Connected Layer (exactly as seen in regular Neural Networks). We -will stack these layers to form a full CNN architecture. +Like all statistical methods, supervised learning using neural +networks has important limitations. This is especially important when +one seeks to apply these methods, especially to physics problems. Like +all tools, DNNs are not a universal solution. Often, the same or +better performance on a task can be achieved by using a few +hand-engineered features (or even a collection of random +features).

    -A simple CNN for image classification could have the architecture: +Here we list some of the important limitations of supervised neural network based models.

      -
    • INPUT (\( 32\times 32 \times 3 \)) will hold the raw pixel values of the image, in this case an image of width 32, height 32, and with three color channels R,G,B.
    • -
    • CONV (convolutional )layer will compute the output of neurons that are connected to local regions in the input, each computing a dot product between their weights and a small region they are connected to in the input volume. This may result in volume such as \( [32\times 32\times 12] \) if we decided to use 12 filters.
    • -
    • RELU layer will apply an elementwise activation function, such as the \( max(0,x) \) thresholding at zero. This leaves the size of the volume unchanged (\( [32\times 32\times 12] \)).
    • -
    • POOL (pooling) layer will perform a downsampling operation along the spatial dimensions (width, height), resulting in volume such as \( [16\times 16\times 12] \).
    • -
    • FC (i.e. fully-connected) layer will compute the class scores, resulting in volume of size \( [1\times 1\times 10] \), where each of the 10 numbers correspond to a class score, such as among the 10 categories of the MNIST images we considered above . As with ordinary Neural Networks and as the name implies, each neuron in this layer will be connected to all the numbers in the previous volume.
    • +
    • Need labeled data. All supervised learning methods, DNNs for supervised learning require labeled data. Often, labeled data is harder to acquire than unlabeled data (e.g. one must pay for human experts to label images).
    • +
    • Supervised neural networks are extremely data intensive. DNNs are data hungry. They perform best when data is plentiful. This is doubly so for supervised methods where the data must also be labeled. The utility of DNNs is extremely limited if data is hard to acquire or the datasets are small (hundreds to a few thousand samples). In this case, the performance of other methods that utilize hand-engineered features can exceed that of DNNs.
    • +
    • Homogeneous data. Almost all DNNs deal with homogeneous data of one type. It is very hard to design architectures that mix and match data types (i.e. some continuous variables, some discrete variables, some time series). In applications beyond images, video, and language, this is often what is required. In contrast, ensemble models like random forests or gradient-boosted trees have no difficulty handling mixed data types.
    • +
    • Many problems are not about prediction. In natural science we are often interested in learning something about the underlying distribution that generates the data. In this case, it is often difficult to cast these ideas in a supervised learning setting. While the problems are related, it is possible to make good predictions with a wrong model. The model might or might not be useful for understanding the underlying science.
    +Some of these remarks are particular to DNNs, others are shared by all supervised learning methods. This motivates the use of unsupervised methods which in part circumvent these problems. + +

      @@ -246,6 +294,15 @@ A simple CNN for image classification could have the architecture:
    • 44
    • 45
    • 46
    • +
    • 47
    • +
    • 48
    • +
    • 49
    • +
    • 50
    • +
    • 51
    • +
    • 52
    • +
    • 53
    • +
    • ...
    • +
    • 62
    • »
    diff --git a/doc/pub/week41/html/._week41-bs044.html b/doc/pub/week41/html/._week41-bs044.html index d25269517..705443614 100644 --- a/doc/pub/week41/html/._week41-bs044.html +++ b/doc/pub/week41/html/._week41-bs044.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,21 +253,41 @@ MathJax.Hub.Config({ -

    Transforming images

    +

    Convolutional Neural Networks (recognizing images)

    -CNNs transform the original image layer by layer from the original -pixel values to the final class scores. +Convolutional neural networks (CNNs) were developed during the last +decade of the previous century, with a focus on character recognition +tasks. Nowadays, CNNs are a central element in the spectacular success +of deep learning methods. The success in for example image +classifications have made them a central tool for most machine +learning practitioners.

    -Observe that some layers contain -parameters and other don’t. In particular, the CNN layers perform -transformations that are a function of not only the activations in the -input volume, but also of the parameters (the weights and biases of -the neurons). On the other hand, the RELU/POOL layers will implement a -fixed function. The parameters in the CONV/FC layers will be trained -with gradient descent so that the class scores that the CNN computes -are consistent with the labels in the training set for each image. +CNNs are very similar to ordinary Neural Networks. +They are made up of neurons that have learnable weights and +biases. Each neuron receives some inputs, performs a dot product and +optionally follows it with a non-linearity. The whole network still +expresses a single differentiable score function: from the raw image +pixels on one end to class scores at the other. And they still have a +loss function (for example Softmax) on the last (fully-connected) layer +and all the tips/tricks we developed for learning regular Neural +Networks still apply (back propagation, gradient descent etc etc). + +

    +What is the difference? CNN architectures make the explicit assumption that +the inputs are images, which allows us to encode certain properties +into the architecture. These then make the forward function more +efficient to implement and vastly reduce the amount of parameters in +the network. + +

    +Here we provide only a superficial overview, for the more interested, we recommend highly the course +IN5400 – Machine Learning for Image Analysis +and the slides of CS231. + +

    +Another good read is the article here https://arxiv.org/pdf/1603.07285.pdf.

    @@ -241,6 +306,16 @@ are consistent with the labels in the training set for each image.

  • 44
  • 45
  • 46
  • +
  • 47
  • +
  • 48
  • +
  • 49
  • +
  • 50
  • +
  • 51
  • +
  • 52
  • +
  • 53
  • +
  • 54
  • +
  • ...
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs045.html b/doc/pub/week41/html/._week41-bs045.html index 8e307bb28..a099b07ce 100644 --- a/doc/pub/week41/html/._week41-bs045.html +++ b/doc/pub/week41/html/._week41-bs045.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -208,26 +253,32 @@ MathJax.Hub.Config({ -

    CNNs in brief

    +

    Regular NNs don’t scale well to full images

    -In summary: - -

      -
    • A CNN architecture is in the simplest case a list of Layers that transform the image volume into an output volume (e.g. holding the class scores)
    • -
    • There are a few distinct types of Layers (e.g. CONV/FC/RELU/POOL are by far the most popular)
    • -
    • Each Layer accepts an input 3D volume and transforms it to an output 3D volume through a differentiable function
    • -
    • Each Layer may or may not have parameters (e.g. CONV/FC do, RELU/POOL don’t)
    • -
    • Each Layer may or may not have additional hyperparameters (e.g. CONV/FC/POOL do, RELU doesn’t)
    • -
    - -For more material on convolutional networks, we strongly recommend -the course -IN5400 – Machine Learning for Image Analysis -and the slides of CS231 which is taught at Stanford University (consistently ranked as one of the top computer science programs in the world). Michael Nielsen's book is a must read, in particular chapter 6 which deals with CNNs. +As an example, consider +an image of size \( 32\times 32\times 3 \) (32 wide, 32 high, 3 color channels), so a +single fully-connected neuron in a first hidden layer of a regular +Neural Network would have \( 32\times 32\times 3 = 3072 \) weights. This amount still +seems manageable, but clearly this fully-connected structure does not +scale to larger images. For example, an image of more respectable +size, say \( 200\times 200\times 3 \), would lead to neurons that have +\( 200\times 200\times 3 = 120,000 \) weights.

    +We could have +several such neurons, and the parameters would add up quickly! Clearly, +this full connectivity is wasteful and the huge number of parameters +would quickly lead to possible overfitting. +

    +

    +
    +

    Figure 1: A regular 3-layer Neural Network.

    +

    +
    + +

    diff --git a/doc/pub/week41/html/._week41-bs046.html b/doc/pub/week41/html/._week41-bs046.html index a7a4d3d8d..055bae73b 100644 --- a/doc/pub/week41/html/._week41-bs046.html +++ b/doc/pub/week41/html/._week41-bs046.html @@ -41,50 +41,56 @@ Automatically generated HTML file from DocOnce source @@ -149,52 +177,67 @@ MathJax.Hub.Config({ @@ -210,26 +253,44 @@ MathJax.Hub.Config({ -

    CNNs in brief

    +

    3D volumes of neurons

    -In summary: - -

      -
    • A CNN architecture is in the simplest case a list of Layers that transform the image volume into an output volume (e.g. holding the class scores)
    • -
    • There are a few distinct types of Layers (e.g. CONV/FC/RELU/POOL are by far the most popular)
    • -
    • Each Layer accepts an input 3D volume and transforms it to an output 3D volume through a differentiable function
    • -
    • Each Layer may or may not have parameters (e.g. CONV/FC do, RELU/POOL don’t)
    • -
    • Each Layer may or may not have additional hyperparameters (e.g. CONV/FC/POOL do, RELU doesn’t)
    • -
    - -For more material on convolutional networks, we strongly recommend -the course -IN5400 – Machine Learning for Image Analysis -and the slides of CS231 which is taught at Stanford University (consistently ranked as one of the top computer science programs in the world). Michael Nielsen's book is a must read, in particular chapter 6 which deals with CNNs. +Convolutional Neural Networks take advantage of the fact that the +input consists of images and they constrain the architecture in a more +sensible way.

    +In particular, unlike a regular Neural Network, the +layers of a CNN have neurons arranged in 3 dimensions: width, +height, depth. (Note that the word depth here refers to the third +dimension of an activation volume, not to the depth of a full Neural +Network, which can refer to the total number of layers in a network.) +

    +To understand it better, the above example of an image +with an input volume of +activations has dimensions \( 32\times 32\times 3 \) (width, height, +depth respectively). + +

    +The neurons in a layer will +only be connected to a small region of the layer before it, instead of +all of the neurons in a fully-connected manner. Moreover, the final +output layer could for this specific image have dimensions \( 1\times 1 \times 10 \), +because by the +end of the CNN architecture we will reduce the full image into a +single vector of class scores, arranged along the depth +dimension. + +

    +

    +
    +

    Figure 2: A CNN arranges its neurons in three dimensions (width, height, depth), as visualized in one of the layers. Every layer of a CNN transforms the 3D input volume to a 3D output volume of neuron activations. In this example, the red input layer holds the image, so its width and height would be the dimensions of the image, and the depth would be 3 (Red, Green, Blue channels).

    +

    +
    + +

    diff --git a/doc/pub/week41/html/._week41-bs047.html b/doc/pub/week41/html/._week41-bs047.html index 2d846eb0c..496ab1504 100644 --- a/doc/pub/week41/html/._week41-bs047.html +++ b/doc/pub/week41/html/._week41-bs047.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -243,18 +251,29 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Setting it up

    +

    Layers used to build CNNs

    -It means that to represent the entire -dataset of images, we require a 4D matrix or tensor. This tensor has the dimensions: -$$ -(n_{inputs},\, n_{pixels, width},\, n_{pixels, height},\, depth) . -$$ +A simple CNN is a sequence of layers, and every layer of a CNN +transforms one volume of activations to another through a +differentiable function. We use three main types of layers to build +CNN architectures: Convolutional Layer, Pooling Layer, and +Fully-Connected Layer (exactly as seen in regular Neural Networks). We +will stack these layers to form a full CNN architecture.

    +A simple CNN for image classification could have the architecture: + +

      +
    • INPUT (\( 32\times 32 \times 3 \)) will hold the raw pixel values of the image, in this case an image of width 32, height 32, and with three color channels R,G,B.
    • +
    • CONV (convolutional )layer will compute the output of neurons that are connected to local regions in the input, each computing a dot product between their weights and a small region they are connected to in the input volume. This may result in volume such as \( [32\times 32\times 12] \) if we decided to use 12 filters.
    • +
    • RELU layer will apply an elementwise activation function, such as the \( max(0,x) \) thresholding at zero. This leaves the size of the volume unchanged (\( [32\times 32\times 12] \)).
    • +
    • POOL (pooling) layer will perform a downsampling operation along the spatial dimensions (width, height), resulting in volume such as \( [16\times 16\times 12] \).
    • +
    • FC (i.e. fully-connected) layer will compute the class scores, resulting in volume of size \( [1\times 1\times 10] \), where each of the 10 numbers correspond to a class score, such as among the 10 categories of the MNIST images we considered above . As with ordinary Neural Networks and as the name implies, each neuron in this layer will be connected to all the numbers in the previous volume.
    • +
    +

    diff --git a/doc/pub/week41/html/._week41-bs048.html b/doc/pub/week41/html/._week41-bs048.html index 0fa9e4e1f..3afcbef92 100644 --- a/doc/pub/week41/html/._week41-bs048.html +++ b/doc/pub/week41/html/._week41-bs048.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -245,20 +253,21 @@ MathJax.Hub.Config({ -

    The MNIST dataset again

    +

    Transforming images

    -The MNIST dataset consists of grayscale images with a pixel size of -\( 28\times 28 \), meaning we require \( 28 \times 28 = 724 \) weights to each -neuron in the first hidden layer. +CNNs transform the original image layer by layer from the original +pixel values to the final class scores.

    -If we were to analyze images of size \( 128\times 128 \) we would require -\( 128 \times 128 = 16384 \) weights to each neuron. Even worse if we were -dealing with color images, as most images are, we have an image matrix -of size \( 128\times 128 \) for each color dimension (Red, Green, Blue), -meaning 3 times the number of weights \( = 49152 \) are required for every -single neuron in the first hidden layer. +Observe that some layers contain +parameters and other don’t. In particular, the CNN layers perform +transformations that are a function of not only the activations in the +input volume, but also of the parameters (the weights and biases of +the neurons). On the other hand, the RELU/POOL layers will implement a +fixed function. The parameters in the CONV/FC layers will be trained +with gradient descent so that the class scores that the CNN computes +are consistent with the labels in the training set for each image.

    @@ -286,7 +295,7 @@ single neuron in the first hidden layer.

  • 57
  • 58
  • ...
  • -
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs049.html b/doc/pub/week41/html/._week41-bs049.html index 310eaf909..bc429afcf 100644 --- a/doc/pub/week41/html/._week41-bs049.html +++ b/doc/pub/week41/html/._week41-bs049.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -245,18 +253,23 @@ MathJax.Hub.Config({ -

    Strong correlations

    -Images typically have strong local correlations, meaning that a small -part of the image varies little from its neighboring regions. If for -example we have an image of a blue car, we can roughly assume that a -small blue part of the image is surrounded by other blue regions. +

    CNNs in brief

    -Therefore, instead of connecting every single pixel to a neuron in the -first hidden layer, as we have previously done with deep neural -networks, we can instead connect each neuron to a small part of the -image (in all 3 RGB depth dimensions). The size of each small area is -fixed, and known as a receptive. +In summary: + +

      +
    • A CNN architecture is in the simplest case a list of Layers that transform the image volume into an output volume (e.g. holding the class scores)
    • +
    • There are a few distinct types of Layers (e.g. CONV/FC/RELU/POOL are by far the most popular)
    • +
    • Each Layer accepts an input 3D volume and transforms it to an output 3D volume through a differentiable function
    • +
    • Each Layer may or may not have parameters (e.g. CONV/FC do, RELU/POOL don’t)
    • +
    • Each Layer may or may not have additional hyperparameters (e.g. CONV/FC/POOL do, RELU doesn’t)
    • +
    + +For more material on convolutional networks, we strongly recommend +the course +IN5400 – Machine Learning for Image Analysis +and the slides of CS231 which is taught at Stanford University (consistently ranked as one of the top computer science programs in the world). Michael Nielsen's book is a must read, in particular chapter 6 which deals with CNNs.

    @@ -284,7 +297,7 @@ fixed, and known as a 58

  • 59
  • ...
  • -
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs050.html b/doc/pub/week41/html/._week41-bs050.html index 151b55eed..0acdcfd37 100644 --- a/doc/pub/week41/html/._week41-bs050.html +++ b/doc/pub/week41/html/._week41-bs050.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -243,26 +251,20 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Layers of a CNN

    -The layers of a convolutional neural network arrange neurons in 3D: width, height and depth. -The input image is typically a square matrix of depth 3. +

    CNNs in more detail, building convolutional neural networks in Tensorflow and Keras

    -A convolution is performed on the image which outputs -a 3D volume of neurons. The weights to the input are arranged in a number of 2D matrices, known as filters. +As discussed above, CNNs are neural networks built from the assumption that the inputs +to the network are 2D images. This is important because the number of features or pixels in images +grows very fast with the image size, and an enormous number of weights and biases are needed in order to build an accurate network.

    -Each filter slides along the input image, taking the dot product -between each small part of the image and the filter, in all depth -dimensions. This is then passed through a non-linear function, -typically the Rectified Linear (ReLu) function, which serves as the -activation of the neurons in the first convolutional layer. This is -further passed through a pooling layer, which reduces the size of the -convolutional layer, e.g. by taking the maximum or average across some -small regions, and this serves as input to the next convolutional -layer. +As before, we still have our input, a hidden layer and an output. What's novel about convolutional networks +are the convolutional and pooling layers stacked in pairs between the input and the hidden layer. +In addition, the data is no longer represented as a 2D feature matrix, instead each input is a number of 2D +matrices, typically 1 for each color dimension (Red, Green, Blue).

    @@ -290,7 +292,7 @@ layer.

  • 59
  • 60
  • ...
  • -
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs051.html b/doc/pub/week41/html/._week41-bs051.html index 7927fbc94..105cabd28 100644 --- a/doc/pub/week41/html/._week41-bs051.html +++ b/doc/pub/week41/html/._week41-bs051.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -245,17 +253,14 @@ MathJax.Hub.Config({ -

    Systematic reduction

    +

    Setting it up

    -By systematically reducing the size of the input volume, through -convolution and pooling, the network should create representations of -small parts of the input, and then from them assemble representations -of larger areas. The final pooling layer is flattened to serve as -input to a hidden layer, such that each neuron in the final pooling -layer is connected to every single neuron in the hidden layer. This -then serves as input to the output layer, e.g. a softmax output for -classification. +It means that to represent the entire +dataset of images, we require a 4D matrix or tensor. This tensor has the dimensions: +$$ +(n_{inputs},\, n_{pixels, width},\, n_{pixels, height},\, depth) . +$$

    @@ -282,6 +287,8 @@ classification.

  • 59
  • 60
  • 61
  • +
  • ...
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs052.html b/doc/pub/week41/html/._week41-bs052.html index 0dd4b8c69..c1d2ceab4 100644 --- a/doc/pub/week41/html/._week41-bs052.html +++ b/doc/pub/week41/html/._week41-bs052.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -245,51 +253,21 @@ MathJax.Hub.Config({ -

    Prerequisites: Collect and pre-process data

    +

    The MNIST dataset again

    +

    +The MNIST dataset consists of grayscale images with a pixel size of +\( 28\times 28 \), meaning we require \( 28 \times 28 = 724 \) weights to each +neuron in the first hidden layer. - -

    # import necessary packages
    -import numpy as np
    -import matplotlib.pyplot as plt
    -from sklearn import datasets
    +

    +If we were to analyze images of size \( 128\times 128 \) we would require +\( 128 \times 128 = 16384 \) weights to each neuron. Even worse if we were +dealing with color images, as most images are, we have an image matrix +of size \( 128\times 128 \) for each color dimension (Red, Green, Blue), +meaning 3 times the number of weights \( = 49152 \) are required for every +single neuron in the first hidden layer. - -# ensure the same random numbers appear every time -np.random.seed(0) - -# display images in notebook -%matplotlib inline -plt.rcParams['figure.figsize'] = (12,12) - - -# download MNIST dataset -digits = datasets.load_digits() - -# define inputs and labels -inputs = digits.images -labels = digits.target - -# RGB images have a depth of 3 -# our images are grayscale so they should have a depth of 1 -inputs = inputs[:,:,:,np.newaxis] - -print("inputs = (n_inputs, pixel_width, pixel_height, depth) = " + str(inputs.shape)) -print("labels = (n_inputs) = " + str(labels.shape)) - - -# choose some random images to display -n_inputs = len(inputs) -indices = np.arange(n_inputs) -random_indices = np.random.choice(indices, size=5) - -for i, image in enumerate(digits.images[random_indices]): - plt.subplot(1, 5, i+1) - plt.axis('off') - plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest') - plt.title("Label: %d" % digits.target[random_indices[i]]) -plt.show() -

    @@ -314,6 +292,7 @@ plt.show()

  • 59
  • 60
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs053.html b/doc/pub/week41/html/._week41-bs053.html index e55212110..edd5cf478 100644 --- a/doc/pub/week41/html/._week41-bs053.html +++ b/doc/pub/week41/html/._week41-bs053.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -245,23 +253,21 @@ MathJax.Hub.Config({ -

    Importing Keras and Tensorflow

    +

    Strong correlations

    +

    +Images typically have strong local correlations, meaning that a small +part of the image varies little from its neighboring regions. If for +example we have an image of a blue car, we can roughly assume that a +small blue part of the image is surrounded by other blue regions. - -

    from keras.utils import to_categorical
    -from sklearn.model_selection import train_test_split
    +

    +Therefore, instead of connecting every single pixel to a neuron in the +first hidden layer, as we have previously done with deep neural +networks, we can instead connect each neuron to a small part of the +image (in all 3 RGB depth dimensions). The size of each small area is +fixed, and known as a receptive. -# representation of labels -labels = to_categorical(labels) - -# split into train and test data -# one-liner from scikit-learn library -train_size = 0.8 -test_size = 1 - train_size -X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size, - test_size=test_size) -

    @@ -285,6 +291,7 @@ X_train, X_test, Y_train, Y_test = train_tes

  • 59
  • 60
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs054.html b/doc/pub/week41/html/._week41-bs054.html index 72baf9c10..7bea87061 100644 --- a/doc/pub/week41/html/._week41-bs054.html +++ b/doc/pub/week41/html/._week41-bs054.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -243,153 +251,27 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Using TensorFlow backend

    +

    Layers of a CNN

    +The layers of a convolutional neural network arrange neurons in 3D: width, height and depth. +The input image is typically a square matrix of depth 3.

    -We need to define model and architecture and choose cost function and optmizer. +A convolution is performed on the image which outputs +a 3D volume of neurons. The weights to the input are arranged in a number of 2D matrices, known as filters. +

    +Each filter slides along the input image, taking the dot product +between each small part of the image and the filter, in all depth +dimensions. This is then passed through a non-linear function, +typically the Rectified Linear (ReLu) function, which serves as the +activation of the neurons in the first convolutional layer. This is +further passed through a pooling layer, which reduces the size of the +convolutional layer, e.g. by taking the maximum or average across some +small regions, and this serves as input to the next convolutional +layer. - -

    import tensorflow as tf
    -
    -class ConvolutionalNeuralNetworkTensorflow:
    -    def __init__(
    -            self,
    -            X_train,
    -            Y_train,
    -            X_test,
    -            Y_test,
    -            n_filters=10,
    -            n_neurons_connected=50,
    -            n_categories=10,
    -            receptive_field=3,
    -            stride=1,
    -            padding=1,
    -            epochs=10,
    -            batch_size=100,
    -            eta=0.1,
    -            lmbd=0.0):
    -        
    -        self.global_step = tf.Variable(0, dtype=tf.int32, trainable=False, name='global_step')
    -        
    -        self.X_train = X_train
    -        self.Y_train = Y_train
    -        self.X_test = X_test
    -        self.Y_test = Y_test
    -        
    -        self.n_inputs, self.input_width, self.input_height, self.depth = X_train.shape
    -        
    -        self.n_filters = n_filters
    -        self.n_downsampled = int(self.input_width*self.input_height*n_filters / 4)
    -        self.n_neurons_connected = n_neurons_connected
    -        self.n_categories = n_categories
    -        
    -        self.receptive_field = receptive_field
    -        self.stride = stride
    -        self.strides = [stride, stride, stride, stride]
    -        self.padding = padding
    -        
    -        self.epochs = epochs
    -        self.batch_size = batch_size
    -        self.iterations = self.n_inputs // self.batch_size
    -        self.eta = eta
    -        self.lmbd = lmbd
    -        
    -        self.create_placeholders()
    -        self.create_CNN()
    -        self.create_loss()
    -        self.create_optimiser()
    -        self.create_accuracy()
    -    
    -    def create_placeholders(self):
    -        with tf.name_scope('data'):
    -            self.X = tf.placeholder(tf.float32, shape=(None, self.input_width, self.input_height, self.depth), name='X_data')
    -            self.Y = tf.placeholder(tf.float32, shape=(None, self.n_categories), name='Y_data')
    -    
    -    def create_CNN(self):
    -        with tf.name_scope('CNN'):
    -            
    -            # Convolutional layer
    -            self.W_conv = self.weight_variable([self.receptive_field, self.receptive_field, self.depth, self.n_filters], name='conv', dtype=tf.float32)
    -            b_conv = self.weight_variable([self.n_filters], name='conv', dtype=tf.float32)
    -            z_conv = tf.nn.conv2d(self.X, self.W_conv, self.strides, padding='SAME', name='conv') + b_conv
    -            a_conv = tf.nn.relu(z_conv)
    -            
    -            # 2x2 max pooling
    -            a_pool = tf.nn.max_pool(a_conv, [1, 2, 2, 1], [1, 2, 2, 1], padding='SAME', name='pool')
    -            
    -            # Fully connected layer
    -            a_pool_flat = tf.reshape(a_pool, [-1, self.n_downsampled])
    -            self.W_fc = self.weight_variable([self.n_downsampled, self.n_neurons_connected], name='fc', dtype=tf.float32)
    -            b_fc = self.bias_variable([self.n_neurons_connected], name='fc', dtype=tf.float32)
    -            a_fc = tf.nn.relu(tf.matmul(a_pool_flat, self.W_fc) + b_fc)
    -            
    -            # Output layer
    -            self.W_out = self.weight_variable([self.n_neurons_connected, self.n_categories], name='out', dtype=tf.float32)
    -            b_out = self.bias_variable([self.n_categories], name='out', dtype=tf.float32)
    -            self.z_out = tf.matmul(a_fc, self.W_out) + b_out
    -    
    -    def create_loss(self):
    -        with tf.name_scope('loss'):
    -            softmax_loss = tf.reduce_mean(tf.nn.softmax_cross_entropy_with_logits_v2(labels=self.Y, logits=self.z_out))
    -            
    -            regularizer_loss_conv = tf.nn.l2_loss(self.W_conv)
    -            regularizer_loss_fc = tf.nn.l2_loss(self.W_fc)
    -            regularizer_loss_out = tf.nn.l2_loss(self.W_out)
    -            regularizer_loss = self.lmbd*(regularizer_loss_conv + regularizer_loss_fc + regularizer_loss_out)
    -            
    -            self.loss = softmax_loss + regularizer_loss
    -
    -    def create_accuracy(self):
    -        with tf.name_scope('accuracy'):
    -            probabilities = tf.nn.softmax(self.z_out)
    -            predictions = tf.argmax(probabilities, 1)
    -            labels = tf.argmax(self.Y, 1)
    -            
    -            correct_predictions = tf.equal(predictions, labels)
    -            correct_predictions = tf.cast(correct_predictions, tf.float32)
    -            self.accuracy = tf.reduce_mean(correct_predictions)
    -    
    -    def create_optimiser(self):
    -        with tf.name_scope('optimizer'):
    -            self.optimizer = tf.train.GradientDescentOptimizer(learning_rate=self.eta).minimize(self.loss, global_step=self.global_step)
    -            
    -    def weight_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.truncated_normal(shape, stddev=0.1)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -    
    -    def bias_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.constant(0.1, shape=shape)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -
    -    def fit(self):
    -        data_indices = np.arange(self.n_inputs)
    -
    -        with tf.Session() as sess:
    -            sess.run(tf.global_variables_initializer())
    -            for i in range(self.epochs):
    -                for j in range(self.iterations):
    -                    chosen_datapoints = np.random.choice(data_indices, size=self.batch_size, replace=False)
    -                    batch_X, batch_Y = self.X_train[chosen_datapoints], self.Y_train[chosen_datapoints]
    -            
    -                    sess.run([CNN.loss, CNN.optimizer],
    -                        feed_dict={CNN.X: batch_X,
    -                                   CNN.Y: batch_Y})
    -                    accuracy = sess.run(CNN.accuracy,
    -                        feed_dict={CNN.X: batch_X,
    -                                   CNN.Y: batch_Y})
    -                    step = sess.run(CNN.global_step)
    -    
    -            self.train_loss, self.train_accuracy = sess.run([CNN.loss, CNN.accuracy],
    -                feed_dict={CNN.X: self.X_train,
    -                           CNN.Y: self.Y_train})
    -        
    -            self.test_loss, self.test_accuracy = sess.run([CNN.loss, CNN.accuracy],
    -                feed_dict={CNN.X: self.X_test,
    -                           CNN.Y: self.Y_test})
    -

    @@ -412,6 +294,7 @@ class ConvolutionalNeuralNetworkTensorflow:

  • 59
  • 60
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs055.html b/doc/pub/week41/html/._week41-bs055.html index 871bbbc79..9ee6b8aef 100644 --- a/doc/pub/week41/html/._week41-bs055.html +++ b/doc/pub/week41/html/._week41-bs055.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -245,38 +253,18 @@ MathJax.Hub.Config({ -

    Train the model

    +

    Systematic reduction

    -We need now to train the model, evaluate it and test its performance on test data, and eventually include hyperparameters. -

    +By systematically reducing the size of the input volume, through +convolution and pooling, the network should create representations of +small parts of the input, and then from them assemble representations +of larger areas. The final pooling layer is flattened to serve as +input to a hidden layer, such that each neuron in the final pooling +layer is connected to every single neuron in the hidden layer. This +then serves as input to the output layer, e.g. a softmax output for +classification. - -

    epochs = 100
    -batch_size = 100
    -n_filters = 10
    -n_neurons_connected = 50
    -n_categories = 10
    -
    -eta_vals = np.logspace(-5, 1, 7)
    -lmbd_vals = np.logspace(-5, 1, 7)
    -CNN_tf = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    -        
    -for i, eta in enumerate(eta_vals):
    -    for j, lmbd in enumerate(lmbd_vals):
    -        CNN = ConvolutionalNeuralNetworkTensorflow(X_train, Y_train, X_test, Y_test,
    -                                      n_filters=n_filters, n_neurons_connected=n_neurons_connected,
    -                                      n_categories=n_categories, epochs=epochs, batch_size=batch_size,
    -                                      eta=eta, lmbd=lmbd)
    -        CNN.fit()
    -        
    -        print("Learning rate = ", eta)
    -        print("Lambda = ", lmbd)
    -        print("Test accuracy: %.3f" % CNN.test_accuracy)
    -        print()
    -            
    -        CNN_tf[i][j] = CNN
    -

    @@ -298,6 +286,7 @@ CNN_tf = np.59

  • 60
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs056.html b/doc/pub/week41/html/._week41-bs056.html index cbc540cc2..01ca8517b 100644 --- a/doc/pub/week41/html/._week41-bs056.html +++ b/doc/pub/week41/html/._week41-bs056.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -245,40 +253,49 @@ MathJax.Hub.Config({ -

    Visualizing the results

    - +

    Prerequisites: Collect and pre-process data

    -

    # visual representation of grid search
    -# uses seaborn heatmap, could probably do this in matplotlib
    -import seaborn as sns
    +
    # import necessary packages
    +import numpy as np
    +import matplotlib.pyplot as plt
    +from sklearn import datasets
     
    -sns.set()
     
    -train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +# ensure the same random numbers appear every time
    +np.random.seed(0)
     
    -for i in range(len(eta_vals)):
    -    for j in range(len(lmbd_vals)):
    -        CNN = CNN_tf[i][j]
    +# display images in notebook
    +%matplotlib inline
    +plt.rcParams['figure.figsize'] = (12,12)
     
    -        train_accuracy[i][j] = CNN.train_accuracy
    -        test_accuracy[i][j] = CNN.test_accuracy
     
    -        
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Training Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    +# download MNIST dataset
    +digits = datasets.load_digits()
     
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Test Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    +# define inputs and labels
    +inputs = digits.images
    +labels = digits.target
    +
    +# RGB images have a depth of 3
    +# our images are grayscale so they should have a depth of 1
    +inputs = inputs[:,:,:,np.newaxis]
    +
    +print("inputs = (n_inputs, pixel_width, pixel_height, depth) = " + str(inputs.shape))
    +print("labels = (n_inputs) = " + str(labels.shape))
    +
    +
    +# choose some random images to display
    +n_inputs = len(inputs)
    +indices = np.arange(n_inputs)
    +random_indices = np.random.choice(indices, size=5)
    +
    +for i, image in enumerate(digits.images[random_indices]):
    +    plt.subplot(1, 5, i+1)
    +    plt.axis('off')
    +    plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest')
    +    plt.title("Label: %d" % digits.target[random_indices[i]])
     plt.show()
     

    @@ -301,6 +318,7 @@ plt.show()

  • 59
  • 60
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs057.html b/doc/pub/week41/html/._week41-bs057.html index 2a97cd726..842615fcd 100644 --- a/doc/pub/week41/html/._week41-bs057.html +++ b/doc/pub/week41/html/._week41-bs057.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -243,47 +251,24 @@ MathJax.Hub.Config({

     

     

     

    - - -

    Running with Keras

    + +

    Importing Keras and Tensorflow

    -

    from keras.models import Sequential
    -from keras.layers.convolutional import Conv2D
    -from keras.layers.convolutional import MaxPooling2D
    -from keras.layers import Flatten
    -from keras.layers import Dense
    -from keras.regularizers import l2
    -from keras.optimizers import SGD
    +
    from keras.utils import to_categorical
    +from sklearn.model_selection import train_test_split
     
    -def create_convolutional_neural_network_keras(input_shape, receptive_field,
    -                                              n_filters, n_neurons_connected, n_categories,
    -                                              eta, lmbd):
    -    model = Sequential()
    -    model.add(Conv2D(n_filters, (receptive_field, receptive_field), input_shape=input_shape, padding='same',
    -              activation='relu', kernel_regularizer=l2(lmbd)))
    -    model.add(MaxPooling2D(pool_size=(2, 2)))
    -    model.add(Flatten())
    -    model.add(Dense(n_neurons_connected, activation='relu', kernel_regularizer=l2(lmbd)))
    -    model.add(Dense(n_categories, activation='softmax', kernel_regularizer=l2(lmbd)))
    -    
    -    sgd = SGD(lr=eta)
    -    model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])
    -    
    -    return model
    +# representation of labels
    +labels = to_categorical(labels)
     
    -epochs = 100
    -batch_size = 100
    -input_shape = X_train.shape[1:4]
    -receptive_field = 3
    -n_filters = 10
    -n_neurons_connected = 50
    -n_categories = 10
    -
    -eta_vals = np.logspace(-5, 1, 7)
    -lmbd_vals = np.logspace(-5, 1, 7)
    +# split into train and test data
    +# one-liner from scikit-learn library
    +train_size = 0.8
    +test_size = 1 - train_size
    +X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,
    +                                                    test_size=test_size)
     

    @@ -304,6 +289,7 @@ lmbd_vals = np.

  • 59
  • 60
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs058.html b/doc/pub/week41/html/._week41-bs058.html index 32273ae2b..bed3fbe8d 100644 --- a/doc/pub/week41/html/._week41-bs058.html +++ b/doc/pub/week41/html/._week41-bs058.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -243,29 +251,47 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Final part

    +

    Running with Keras

    -

    CNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    -        
    -for i, eta in enumerate(eta_vals):
    -    for j, lmbd in enumerate(lmbd_vals):
    -        CNN = create_convolutional_neural_network_keras(input_shape, receptive_field,
    +
    from keras.models import Sequential
    +from keras.layers.convolutional import Conv2D
    +from keras.layers.convolutional import MaxPooling2D
    +from keras.layers import Flatten
    +from keras.layers import Dense
    +from keras.regularizers import l2
    +from keras.optimizers import SGD
    +
    +def create_convolutional_neural_network_keras(input_shape, receptive_field,
                                                   n_filters, n_neurons_connected, n_categories,
    -                                              eta, lmbd)
    -        CNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0)
    -        scores = CNN.evaluate(X_test, Y_test)
    -        
    -        CNN_keras[i][j] = CNN
    -        
    -        print("Learning rate = ", eta)
    -        print("Lambda = ", lmbd)
    -        print("Test accuracy: %.3f" % scores[1])
    -        print()
    +                                              eta, lmbd):
    +    model = Sequential()
    +    model.add(Conv2D(n_filters, (receptive_field, receptive_field), input_shape=input_shape, padding='same',
    +              activation='relu', kernel_regularizer=l2(lmbd)))
    +    model.add(MaxPooling2D(pool_size=(2, 2)))
    +    model.add(Flatten())
    +    model.add(Dense(n_neurons_connected, activation='relu', kernel_regularizer=l2(lmbd)))
    +    model.add(Dense(n_categories, activation='softmax', kernel_regularizer=l2(lmbd)))
    +    
    +    sgd = SGD(lr=eta)
    +    model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])
    +    
    +    return model
    +
    +epochs = 100
    +batch_size = 100
    +input_shape = X_train.shape[1:4]
    +receptive_field = 3
    +n_filters = 10
    +n_neurons_connected = 50
    +n_categories = 10
    +
    +eta_vals = np.logspace(-5, 1, 7)
    +lmbd_vals = np.logspace(-5, 1, 7)
     

    @@ -285,6 +311,7 @@ MathJax.Hub.Config({

  • 59
  • 60
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs059.html b/doc/pub/week41/html/._week41-bs059.html index a40598793..99d9d441d 100644 --- a/doc/pub/week41/html/._week41-bs059.html +++ b/doc/pub/week41/html/._week41-bs059.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -245,41 +253,27 @@ MathJax.Hub.Config({ -

    Final visualization

    +

    Final part

    - -

    # visual representation of grid search
    -# uses seaborn heatmap, could probably do this in matplotlib
    -import seaborn as sns
    -
    -sns.set()
    -
    -train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -
    -for i in range(len(eta_vals)):
    -    for j in range(len(lmbd_vals)):
    -        CNN = CNN_keras[i][j]
    -
    -        train_accuracy[i][j] = CNN.evaluate(X_train, Y_train)[1]
    -        test_accuracy[i][j] = CNN.evaluate(X_test, Y_test)[1]
    -
    +
    +
    CNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
             
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Training Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    -
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Test Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    +for i, eta in enumerate(eta_vals):
    +    for j, lmbd in enumerate(lmbd_vals):
    +        CNN = create_convolutional_neural_network_keras(input_shape, receptive_field,
    +                                              n_filters, n_neurons_connected, n_categories,
    +                                              eta, lmbd)
    +        CNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0)
    +        scores = CNN.evaluate(X_test, Y_test)
    +        
    +        CNN_keras[i][j] = CNN
    +        
    +        print("Learning rate = ", eta)
    +        print("Lambda = ", lmbd)
    +        print("Test accuracy: %.3f" % scores[1])
    +        print()
     

    @@ -298,6 +292,7 @@ plt.show()

  • 59
  • 60
  • 61
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/._week41-bs060.html b/doc/pub/week41/html/._week41-bs060.html index 702f00dbb..8c2fb6676 100644 --- a/doc/pub/week41/html/._week41-bs060.html +++ b/doc/pub/week41/html/._week41-bs060.html @@ -41,98 +41,105 @@ Automatically generated HTML file from DocOnce source @@ -170,66 +177,67 @@ MathJax.Hub.Config({ @@ -245,14 +253,43 @@ MathJax.Hub.Config({ -

    Fun links

    +

    Final visualization

    -
      -
    1. Self-Driving cars using a convolutional neural network
    2. -
    3. Abstract art using convolutional neural networks
    4. -
    +

    + +

    # visual representation of grid search
    +# uses seaborn heatmap, could probably do this in matplotlib
    +import seaborn as sns
     
    +sns.set()
    +
    +train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +
    +for i in range(len(eta_vals)):
    +    for j in range(len(lmbd_vals)):
    +        CNN = CNN_keras[i][j]
    +
    +        train_accuracy[i][j] = CNN.evaluate(X_train, Y_train)[1]
    +        test_accuracy[i][j] = CNN.evaluate(X_test, Y_test)[1]
    +
    +        
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Training Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
    +
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Test Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
    +
    +

      @@ -268,6 +305,8 @@ MathJax.Hub.Config({
    • 59
    • 60
    • 61
    • +
    • 62
    • +
    • »
    diff --git a/doc/pub/week41/html/week41-bs.html b/doc/pub/week41/html/week41-bs.html index 045b39a7d..6c79be947 100644 --- a/doc/pub/week41/html/week41-bs.html +++ b/doc/pub/week41/html/week41-bs.html @@ -78,39 +78,68 @@ Automatically generated HTML file from DocOnce source None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -175,24 +204,40 @@ MathJax.Hub.Config({
  • Visualization
  • Building neural networks in Tensorflow and Keras
  • Tensorflow
  • -
  • Collect and pre-process data
  • -
  • Using TensorFlow backend
  • -
  • Optimizing and using gradient descent
  • -
  • Using Keras
  • -
  • The Breast Cancer Data, now with Keras
  • +
  • Using Keras
  • +
  • Collect and pre-process data
  • +
  • The Breast Cancer Data, now with Keras
  • +
  • Fine-tuning neural network hyperparameters
  • +
  • Hidden layers
  • Which activation function should I use?
  • -
  • Is the Logistic activation function (Sigmoid) our choice?
  • -
  • The derivative of the Logistic funtion
  • -
  • The RELU function family
  • -
  • Which activation function should we use?
  • -
  • A top-down perspective on Neural networks
  • -
  • Limitations of supervised learning with deep networks
  • -
  • Convolutional Neural Networks (recognizing images)
  • -
  • Regular NNs don’t scale well to full images
  • -
  • 3D volumes of neurons
  • -
  • Layers used to build CNNs
  • -
  • Transforming images
  • -
  • CNNs in brief
  • +
  • More on activation functions, output layers
  • +
  • Is the Logistic activation function (Sigmoid) our choice?
  • +
  • The derivative of the Logistic funtion
  • +
  • The RELU function family
  • +
  • Which activation function should we use?
  • +
  • Batch Normalization
  • +
  • Dropout
  • +
  • Gradient Clipping
  • +
  • A top-down perspective on Neural networks
  • +
  • Limitations of supervised learning with deep networks
  • +
  • Convolutional Neural Networks (recognizing images)
  • +
  • Regular NNs don’t scale well to full images
  • +
  • 3D volumes of neurons
  • +
  • Layers used to build CNNs
  • +
  • Transforming images
  • +
  • CNNs in brief
  • +
  • CNNs in more detail, building convolutional neural networks in Tensorflow and Keras
  • +
  • Setting it up
  • +
  • The MNIST dataset again
  • +
  • Strong correlations
  • +
  • Layers of a CNN
  • +
  • Systematic reduction
  • +
  • Prerequisites: Collect and pre-process data
  • +
  • Importing Keras and Tensorflow
  • +
  • Running with Keras
  • +
  • Final part
  • +
  • Final visualization
  • +
  • Fun links
  • @@ -227,7 +272,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 8, 2020

    +

    Oct 9, 2020


    @@ -251,7 +296,7 @@ MathJax.Hub.Config({

  • 9
  • 10
  • ...
  • -
  • 46
  • +
  • 62
  • »
  • diff --git a/doc/pub/week41/html/week41-reveal.html b/doc/pub/week41/html/week41-reveal.html index 05e3442c3..7bb99a7c6 100644 --- a/doc/pub/week41/html/week41-reveal.html +++ b/doc/pub/week41/html/week41-reveal.html @@ -148,7 +148,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

     
    -

    Oct 8, 2020

    +

    Oct 9, 2020


    @@ -1510,16 +1510,52 @@ To install tensorflow on Unix/Linux systems, use pip as

    and/or if you use anaconda, just write (or install from the graphical user interface) +(current release of CPU-only TensorFlow)

    -

    conda install tensorflow
    +
    conda create -n tf tensorflow
    +conda activate tf
    +
    +

    +To install the current release of GPU TensorFlow +

    + + +

    conda create -n tf-gpu tensorflow-gpu
    +conda activate tf-gpu
     
    -

    Collect and pre-process data

    +

    Using Keras

    + +

    +Keras is a high level neural network +that supports Tensorflow, CTNK and Theano as backends. +If you have Tensorflow installed Keras is available through the tf.keras module. +If you have Anaconda installed you may run the following command +

    + + +

    conda install keras
    +
    +

    +Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager: + +

    + + +

    pip install keras
    +
    +

    +or look up the instructions here. +

    + + +
    +

    Collect and pre-process data

    @@ -1527,6 +1563,7 @@ and/or if you use anaconda, just write (or install from the graphical use

    # import necessary packages
     import numpy as np
     import matplotlib.pyplot as plt
    +import tensorflow as tf
     from sklearn import datasets
     
     
    @@ -1570,7 +1607,13 @@ plt.show()
     

    -

    from keras.utils import to_categorical
    +
    from tensorflow.keras.layers import Input
    +from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    +from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    +from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    +from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    +from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    +
     from sklearn.model_selection import train_test_split
     
     # one-hot representation of labels
    @@ -1582,268 +1625,10 @@ test_size = 1 - train_size
     X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,
                                                         test_size=test_size)
     
    -
    - - -
    -

    Using TensorFlow backend

    - -
      -

    1. Define model and architecture
    2. -

    3. Choose cost function and optimizer
    4. -

    -

    import tensorflow as tf
    -
    -class NeuralNetworkTensorflow:
    -    def __init__(
    -            self,
    -            X_train,
    -            Y_train,
    -            X_test,
    -            Y_test,
    -            n_neurons_layer1=100,
    -            n_neurons_layer2=50,
    -            n_categories=2,
    -            epochs=10,
    -            batch_size=100,
    -            eta=0.1,
    -            lmbd=0.0):
    -        
    -        # keep track of number of steps
    -        self.global_step = tf.Variable(0, dtype=tf.int32, trainable=False, name='global_step')
    -        
    -        self.X_train = X_train
    -        self.Y_train = Y_train
    -        self.X_test = X_test
    -        self.Y_test = Y_test
    -        
    -        self.n_inputs = X_train.shape[0]
    -        self.n_features = X_train.shape[1]
    -        self.n_neurons_layer1 = n_neurons_layer1
    -        self.n_neurons_layer2 = n_neurons_layer2
    -        self.n_categories = n_categories
    -        
    -        self.epochs = epochs
    -        self.batch_size = batch_size
    -        self.iterations = self.n_inputs // self.batch_size
    -        self.eta = eta
    -        self.lmbd = lmbd
    -        
    -        # build network piece by piece
    -        # name scopes (with) are used to enforce creation of new variables
    -        # https://www.tensorflow.org/guide/variables
    -        self.create_placeholders()
    -        self.create_DNN()
    -        self.create_loss()
    -        self.create_optimiser()
    -        self.create_accuracy()
    -    
    -    def create_placeholders(self):
    -        # placeholders are fine here, but "Datasets" are the preferred method
    -        # of streaming data into a model
    -        with tf.name_scope('data'):
    -            self.X = tf.placeholder(tf.float32, shape=(None, self.n_features), name='X_data')
    -            self.Y = tf.placeholder(tf.float32, shape=(None, self.n_categories), name='Y_data')
    -    
    -    def create_DNN(self):
    -        with tf.name_scope('DNN'):
    -            # the weights are stored to calculate regularization loss later
    -            
    -            # Fully connected layer 1
    -            self.W_fc1 = self.weight_variable([self.n_features, self.n_neurons_layer1], name='fc1', dtype=tf.float32)
    -            b_fc1 = self.bias_variable([self.n_neurons_layer1], name='fc1', dtype=tf.float32)
    -            a_fc1 = tf.nn.sigmoid(tf.matmul(self.X, self.W_fc1) + b_fc1)
    -            
    -            # Fully connected layer 2
    -            self.W_fc2 = self.weight_variable([self.n_neurons_layer1, self.n_neurons_layer2], name='fc2', dtype=tf.float32)
    -            b_fc2 = self.bias_variable([self.n_neurons_layer2], name='fc2', dtype=tf.float32)
    -            a_fc2 = tf.nn.sigmoid(tf.matmul(a_fc1, self.W_fc2) + b_fc2)
    -            
    -            # Output layer
    -            self.W_out = self.weight_variable([self.n_neurons_layer2, self.n_categories], name='out', dtype=tf.float32)
    -            b_out = self.bias_variable([self.n_categories], name='out', dtype=tf.float32)
    -            self.z_out = tf.matmul(a_fc2, self.W_out) + b_out
    -    
    -    def create_loss(self):
    -        with tf.name_scope('loss'):
    -            softmax_loss = tf.reduce_mean(tf.nn.softmax_cross_entropy_with_logits_v2(labels=self.Y, logits=self.z_out))
    -            
    -            regularizer_loss_fc1 = tf.nn.l2_loss(self.W_fc1)
    -            regularizer_loss_fc2 = tf.nn.l2_loss(self.W_fc2)
    -            regularizer_loss_out = tf.nn.l2_loss(self.W_out)
    -            regularizer_loss = self.lmbd*(regularizer_loss_fc1 + regularizer_loss_fc2 + regularizer_loss_out)
    -            
    -            self.loss = softmax_loss + regularizer_loss
    -
    -    def create_accuracy(self):
    -        with tf.name_scope('accuracy'):
    -            probabilities = tf.nn.softmax(self.z_out)
    -            predictions = tf.argmax(probabilities, axis=1)
    -            labels = tf.argmax(self.Y, axis=1)
    -            
    -            correct_predictions = tf.equal(predictions, labels)
    -            correct_predictions = tf.cast(correct_predictions, tf.float32)
    -            self.accuracy = tf.reduce_mean(correct_predictions)
    -    
    -    def create_optimiser(self):
    -        with tf.name_scope('optimizer'):
    -            self.optimizer = tf.train.GradientDescentOptimizer(learning_rate=self.eta).minimize(self.loss, global_step=self.global_step)
    -            
    -    def weight_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.truncated_normal(shape, stddev=0.1)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -    
    -    def bias_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.constant(0.1, shape=shape)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -    
    -    def fit(self):
    -        data_indices = np.arange(self.n_inputs)
    -
    -        with tf.Session() as sess:
    -            sess.run(tf.global_variables_initializer())
    -            for i in range(self.epochs):
    -                for j in range(self.iterations):
    -                    chosen_datapoints = np.random.choice(data_indices, size=self.batch_size, replace=False)
    -                    batch_X, batch_Y = self.X_train[chosen_datapoints], self.Y_train[chosen_datapoints]
    -            
    -                    sess.run([DNN.loss, DNN.optimizer],
    -                        feed_dict={DNN.X: batch_X,
    -                                   DNN.Y: batch_Y})
    -                    accuracy = sess.run(DNN.accuracy,
    -                        feed_dict={DNN.X: batch_X,
    -                                   DNN.Y: batch_Y})
    -                    step = sess.run(DNN.global_step)
    -    
    -            self.train_loss, self.train_accuracy = sess.run([DNN.loss, DNN.accuracy],
    -                feed_dict={DNN.X: self.X_train,
    -                           DNN.Y: self.Y_train})
    -        
    -            self.test_loss, self.test_accuracy = sess.run([DNN.loss, DNN.accuracy],
    -                feed_dict={DNN.X: self.X_test,
    -                           DNN.Y: self.Y_test})
    -
    -
    - - -
    -

    Optimizing and using gradient descent

    - -

    - - -

    epochs = 100
    -batch_size = 100
    -n_neurons_layer1 = 100
    -n_neurons_layer2 = 50
    -n_categories = 10
    -eta_vals = np.logspace(-5, 1, 7)
    -lmbd_vals = np.logspace(-5, 1, 7)
    -
    -

    - - -

    DNN_tf = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    -        
    -for i, eta in enumerate(eta_vals):
    -    for j, lmbd in enumerate(lmbd_vals):
    -        DNN = NeuralNetworkTensorflow(X_train, Y_train, X_test, Y_test,
    -                                      n_neurons_layer1, n_neurons_layer2, n_categories,
    -                                      epochs=epochs, batch_size=batch_size, eta=eta, lmbd=lmbd)
    -        DNN.fit()
    -        
    -        DNN_tf[i][j] = DNN
    -        
    -        print("Learning rate = ", eta)
    -        print("Lambda = ", lmbd)
    -        print("Test accuracy: %.3f" % DNN.test_accuracy)
    -        print()
    -
    -

    - - -

    # optional
    -# visual representation of grid search
    -# uses seaborn heatmap, could probably do this in matplotlib
    -import seaborn as sns
    -
    -sns.set()
    -
    -train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -
    -for i in range(len(eta_vals)):
    -    for j in range(len(lmbd_vals)):
    -        DNN = DNN_tf[i][j]
    -
    -        train_accuracy[i][j] = DNN.train_accuracy
    -        test_accuracy[i][j] = DNN.test_accuracy
    -
    -        
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Training Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    -
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Test Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    -
    -

    - - -

    # optional
    -# we can use log files to visualize our graph in Tensorboard
    -writer = tf.summary.FileWriter('logs/')
    -writer.add_graph(tf.get_default_graph())
    -
    -
    - - -
    -

    Using Keras

    - -

    -Keras is a high level neural network -that supports Tensorflow, CTNK and Theano as backends. -If you have Tensorflow installed Keras is available through the tf.keras module. -If you have Anaconda installed you may run the following command -

    - - -

    conda install keras
    -
    -

    -Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager: - -

    - - -

    pip install keras
    -
    -

    -or look up the instructions here. - -

    - - -

    import tensorflow as tf
    -from tensorflow.keras.layers import Input
    -from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    -from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    -from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    -from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    -from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    -
    -def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):
    +
    def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):
         model = Sequential()
         model.add(Dense(n_neurons_layer1, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
         model.add(Dense(n_neurons_layer2, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
    @@ -1912,7 +1697,7 @@ plt.show()
     
     
     
    -

    The Breast Cancer Data, now with Keras

    +

    The Breast Cancer Data, now with Keras

    @@ -2086,6 +1871,54 @@ plot_data(eta,n_neuron,Test_accuracy, 'testing&

    +
    +

    Fine-tuning neural network hyperparameters

    + +

    +The flexibility of neural networks is also one of their main +drawbacks: there are many hyperparameters to tweak. Not only can you +use any imaginable network topology (how neurons/nodes are interconnected), +but even in a simple FFNN you can change the number of layers, the +number of neurons per layer, the type of activation function to use in +each layer, the weight initialization logic, the stochastic gradient optmized and much more. How do you +know what combination of hyperparameters is the best for your task? + +

      +

    • You can use grid search with cross-validation to find the right hyperparameters.
    • +
    +

    + +However,since there are many hyperparameters to tune, and since +training a neural network on a large dataset takes a lot of time, you +will only be able to explore a tiny part of the hyperparameter space. + +

      +

    • You can use randomized search.
    • +

    • Or use tools like Oscar, which implements more complex algorithms to help you find a good set of hyperparameters quickly.
    • +
    +
    + + +
    +

    Hidden layers

    + +

    +For many problems you can start with just one or two hidden layers and it will work just fine. +For the MNIST data set you ca easily get a high accuracy using just one hidden layer with a +few hundred neurons. +You can reach for this data set above 98% accuracy using two hidden layers with the same total amount of +neurons, in roughly the same amount of training time. + +

    +For more complex problems, you can gradually +ramp up the number of hidden layers, until you start overfitting the training set. Very complex tasks, such +as large image classification or speech recognition, typically require networks with dozens of layers +and they need a huge amount +of training data. However, you will rarely have to train such networks from scratch: it is much more +common to reuse parts of a pretrained state-of-the-art network that performs a similar task. +

    + +

    Which activation function should I use?

    @@ -2116,7 +1949,26 @@ learn at widely different speeds
    -

    Is the Logistic activation function (Sigmoid) our choice?

    +

    More on activation functions, output layers

    + +

    +In most cases you can use the ReLU activation function in the hidden layers (or one of its variants). + +

    +It is a bit faster to compute than other activation functions, and the gradient descent optimization does in general not get stuck. + +

    +For the output layer: + +

      +

    • For classification the softmax activation function is generally a good choice for classification tasks (when the classes are mutually exclusive).
    • +

    • For regression tasks, you can simply use no activation function at all.
    • +
    +
    + + +
    +

    Is the Logistic activation function (Sigmoid) our choice?

    Although this unfortunate behavior has been empirically observed for @@ -2146,7 +1998,7 @@ better than the logistic function in deep networks).

    -

    The derivative of the Logistic funtion

    +

    The derivative of the Logistic funtion

    Looking at the logistic activation function, when inputs become large @@ -2182,7 +2034,7 @@ fast to compute).

    -

    The RELU function family

    +

    The RELU function family

    The ReLU activation function suffers from a problem known as the dying @@ -2211,7 +2063,7 @@ $$

    -

    Which activation function should we use?

    +

    Which activation function should we use?

    In general it seems that the ELU activation function is better than @@ -2231,7 +2083,58 @@ bootstrap to evaluate other activation functions.

    -

    A top-down perspective on Neural networks

    +

    Batch Normalization

    + +

    +Batch Normalization +aims to address the vanishing/exploding gradients problems, and more generally the problem that the +distribution of each layer’s inputs changes during training, as the parameters of the previous layers change. + +

    +The technique consists of adding an operation in the model just before the activation function of each +layer, simply zero-centering and normalizing the inputs, then scaling and shifting the result using two new +parameters per layer (one for scaling, the other for shifting). In other words, this operation lets the model +learn the optimal scale and mean of the inputs for each layer. +In order to zero-center and normalize the inputs, the algorithm needs to estimate the inputs’ mean and +standard deviation. It does so by evaluating the mean and standard deviation of the inputs over the current +mini-batch, from this the name batch normalization. +

    + + +
    +

    Dropout

    + +

    +It is a fairly simple algorithm: at every training step, every neuron (including the input neurons but +excluding the output neurons) has a probability \( p \) of being temporarily dropped out, meaning it will be +entirely ignored during this training step, but it may be active during the next step. + +

    +The +hyperparameter \( p \) is called the dropout rate, and it is typically set to 50%. After training, the neurons are not dropped anymore. + It is viewed as one of the most popular regularization techniques. +

    + + +
    +

    Gradient Clipping

    + +

    +A popular technique to lessen the exploding gradients problem is to simply clip the gradients during +backpropagation so that they never exceed some threshold (this is mostly useful for recurrent neural +networks). + +

    +This technique is called Gradient Clipping. + +

    +In general however, Batch +Normalization is preferred. +

    + + +
    +

    A top-down perspective on Neural networks

    The first thing we would like to do is divide the data into two or three @@ -2274,7 +2177,7 @@ supervised learning.

    -

    Limitations of supervised learning with deep networks

    +

    Limitations of supervised learning with deep networks

    Like all statistical methods, supervised learning using neural @@ -2301,7 +2204,7 @@ Some of these remarks are particular to DNNs, others are shared by all supervise

    -

    Convolutional Neural Networks (recognizing images)

    +

    Convolutional Neural Networks (recognizing images)

    Convolutional neural networks (CNNs) were developed during the last @@ -2340,7 +2243,7 @@ Another good read is the article here Regular NNs don’t scale well to full images +

    Regular NNs don’t scale well to full images

    As an example, consider @@ -2368,7 +2271,7 @@ would quickly lead to possible overfitting.

    -

    3D volumes of neurons

    +

    3D volumes of neurons

    Convolutional Neural Networks take advantage of the fact that the @@ -2408,7 +2311,7 @@ dimension.

    -

    Layers used to build CNNs

    +

    Layers used to build CNNs

    A simple CNN is a sequence of layers, and every layer of a CNN @@ -2432,7 +2335,7 @@ A simple CNN for image classification could have the architecture:

    -

    Transforming images

    +

    Transforming images

    CNNs transform the original image layer by layer from the original @@ -2451,7 +2354,7 @@ are consistent with the labels in the training set for each image.

    -

    CNNs in brief

    +

    CNNs in brief

    In summary: @@ -2472,6 +2375,300 @@ and the slides of +

    +

    CNNs in more detail, building convolutional neural networks in Tensorflow and Keras

    + +

    +As discussed above, CNNs are neural networks built from the assumption that the inputs +to the network are 2D images. This is important because the number of features or pixels in images +grows very fast with the image size, and an enormous number of weights and biases are needed in order to build an accurate network. + +

    +As before, we still have our input, a hidden layer and an output. What's novel about convolutional networks +are the convolutional and pooling layers stacked in pairs between the input and the hidden layer. +In addition, the data is no longer represented as a 2D feature matrix, instead each input is a number of 2D +matrices, typically 1 for each color dimension (Red, Green, Blue). +

    + + +
    +

    Setting it up

    + +

    +It means that to represent the entire +dataset of images, we require a 4D matrix or tensor. This tensor has the dimensions: + +

     
    +$$ +(n_{inputs},\, n_{pixels, width},\, n_{pixels, height},\, depth) . +$$ +

     
    +

    + + +
    +

    The MNIST dataset again

    + +

    +The MNIST dataset consists of grayscale images with a pixel size of +\( 28\times 28 \), meaning we require \( 28 \times 28 = 724 \) weights to each +neuron in the first hidden layer. + +

    +If we were to analyze images of size \( 128\times 128 \) we would require +\( 128 \times 128 = 16384 \) weights to each neuron. Even worse if we were +dealing with color images, as most images are, we have an image matrix +of size \( 128\times 128 \) for each color dimension (Red, Green, Blue), +meaning 3 times the number of weights \( = 49152 \) are required for every +single neuron in the first hidden layer. +

    + + +
    +

    Strong correlations

    + +

    +Images typically have strong local correlations, meaning that a small +part of the image varies little from its neighboring regions. If for +example we have an image of a blue car, we can roughly assume that a +small blue part of the image is surrounded by other blue regions. + +

    +Therefore, instead of connecting every single pixel to a neuron in the +first hidden layer, as we have previously done with deep neural +networks, we can instead connect each neuron to a small part of the +image (in all 3 RGB depth dimensions). The size of each small area is +fixed, and known as a receptive. +

    + + +
    +

    Layers of a CNN

    +The layers of a convolutional neural network arrange neurons in 3D: width, height and depth. +The input image is typically a square matrix of depth 3. + +

    +A convolution is performed on the image which outputs +a 3D volume of neurons. The weights to the input are arranged in a number of 2D matrices, known as filters. + +

    +Each filter slides along the input image, taking the dot product +between each small part of the image and the filter, in all depth +dimensions. This is then passed through a non-linear function, +typically the Rectified Linear (ReLu) function, which serves as the +activation of the neurons in the first convolutional layer. This is +further passed through a pooling layer, which reduces the size of the +convolutional layer, e.g. by taking the maximum or average across some +small regions, and this serves as input to the next convolutional +layer. +

    + + +
    +

    Systematic reduction

    + +

    +By systematically reducing the size of the input volume, through +convolution and pooling, the network should create representations of +small parts of the input, and then from them assemble representations +of larger areas. The final pooling layer is flattened to serve as +input to a hidden layer, such that each neuron in the final pooling +layer is connected to every single neuron in the hidden layer. This +then serves as input to the output layer, e.g. a softmax output for +classification. +

    + + +
    +

    Prerequisites: Collect and pre-process data

    +

    + + +

    # import necessary packages
    +import numpy as np
    +import matplotlib.pyplot as plt
    +from sklearn import datasets
    +
    +
    +# ensure the same random numbers appear every time
    +np.random.seed(0)
    +
    +# display images in notebook
    +%matplotlib inline
    +plt.rcParams['figure.figsize'] = (12,12)
    +
    +
    +# download MNIST dataset
    +digits = datasets.load_digits()
    +
    +# define inputs and labels
    +inputs = digits.images
    +labels = digits.target
    +
    +# RGB images have a depth of 3
    +# our images are grayscale so they should have a depth of 1
    +inputs = inputs[:,:,:,np.newaxis]
    +
    +print("inputs = (n_inputs, pixel_width, pixel_height, depth) = " + str(inputs.shape))
    +print("labels = (n_inputs) = " + str(labels.shape))
    +
    +
    +# choose some random images to display
    +n_inputs = len(inputs)
    +indices = np.arange(n_inputs)
    +random_indices = np.random.choice(indices, size=5)
    +
    +for i, image in enumerate(digits.images[random_indices]):
    +    plt.subplot(1, 5, i+1)
    +    plt.axis('off')
    +    plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest')
    +    plt.title("Label: %d" % digits.target[random_indices[i]])
    +plt.show()
    +
    +
    + + +
    +

    Importing Keras and Tensorflow

    +

    + + +

    from keras.utils import to_categorical
    +from sklearn.model_selection import train_test_split
    +
    +# representation of labels
    +labels = to_categorical(labels)
    +
    +# split into train and test data
    +# one-liner from scikit-learn library
    +train_size = 0.8
    +test_size = 1 - train_size
    +X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,
    +                                                    test_size=test_size)
    +
    +
    + + +
    +

    Running with Keras

    + +

    + + +

    from keras.models import Sequential
    +from keras.layers.convolutional import Conv2D
    +from keras.layers.convolutional import MaxPooling2D
    +from keras.layers import Flatten
    +from keras.layers import Dense
    +from keras.regularizers import l2
    +from keras.optimizers import SGD
    +
    +def create_convolutional_neural_network_keras(input_shape, receptive_field,
    +                                              n_filters, n_neurons_connected, n_categories,
    +                                              eta, lmbd):
    +    model = Sequential()
    +    model.add(Conv2D(n_filters, (receptive_field, receptive_field), input_shape=input_shape, padding='same',
    +              activation='relu', kernel_regularizer=l2(lmbd)))
    +    model.add(MaxPooling2D(pool_size=(2, 2)))
    +    model.add(Flatten())
    +    model.add(Dense(n_neurons_connected, activation='relu', kernel_regularizer=l2(lmbd)))
    +    model.add(Dense(n_categories, activation='softmax', kernel_regularizer=l2(lmbd)))
    +    
    +    sgd = SGD(lr=eta)
    +    model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])
    +    
    +    return model
    +
    +epochs = 100
    +batch_size = 100
    +input_shape = X_train.shape[1:4]
    +receptive_field = 3
    +n_filters = 10
    +n_neurons_connected = 50
    +n_categories = 10
    +
    +eta_vals = np.logspace(-5, 1, 7)
    +lmbd_vals = np.logspace(-5, 1, 7)
    +
    +
    + + +
    +

    Final part

    + +

    + + +

    CNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    +        
    +for i, eta in enumerate(eta_vals):
    +    for j, lmbd in enumerate(lmbd_vals):
    +        CNN = create_convolutional_neural_network_keras(input_shape, receptive_field,
    +                                              n_filters, n_neurons_connected, n_categories,
    +                                              eta, lmbd)
    +        CNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0)
    +        scores = CNN.evaluate(X_test, Y_test)
    +        
    +        CNN_keras[i][j] = CNN
    +        
    +        print("Learning rate = ", eta)
    +        print("Lambda = ", lmbd)
    +        print("Test accuracy: %.3f" % scores[1])
    +        print()
    +
    +
    + + +
    +

    Final visualization

    + +

    + + +

    # visual representation of grid search
    +# uses seaborn heatmap, could probably do this in matplotlib
    +import seaborn as sns
    +
    +sns.set()
    +
    +train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +
    +for i in range(len(eta_vals)):
    +    for j in range(len(lmbd_vals)):
    +        CNN = CNN_keras[i][j]
    +
    +        train_accuracy[i][j] = CNN.evaluate(X_train, Y_train)[1]
    +        test_accuracy[i][j] = CNN.evaluate(X_test, Y_test)[1]
    +
    +        
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Training Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
    +
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Test Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
    +
    +
    + + +
    +

    Fun links

    + +
      +

    1. Self-Driving cars using a convolutional neural network
    2. +

    3. Abstract art using convolutional neural networks
    4. +
    +
    + +
    diff --git a/doc/pub/week41/html/week41-solarized.html b/doc/pub/week41/html/week41-solarized.html index 7e5db69be..b0f026995 100644 --- a/doc/pub/week41/html/week41-solarized.html +++ b/doc/pub/week41/html/week41-solarized.html @@ -98,39 +98,68 @@ div { text-align: justify; text-justify: inter-word; } None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -172,7 +201,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 8, 2020

    +

    Oct 9, 2020












    @@ -1411,15 +1440,51 @@ To install tensorflow on Unix/Linux systems, use pip as

    and/or if you use anaconda, just write (or install from the graphical user interface) +(current release of CPU-only TensorFlow)

    -

    conda install tensorflow
    +
    conda create -n tf tensorflow
    +conda activate tf
    +
    +

    +To install the current release of GPU TensorFlow +

    + + +

    conda create -n tf-gpu tensorflow-gpu
    +conda activate tf-gpu
     











    -

    Collect and pre-process data

    +

    Using Keras

    + +

    +Keras is a high level neural network +that supports Tensorflow, CTNK and Theano as backends. +If you have Tensorflow installed Keras is available through the tf.keras module. +If you have Anaconda installed you may run the following command +

    + + +

    conda install keras
    +
    +

    +Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager: + +

    + + +

    pip install keras
    +
    +

    +or look up the instructions here. + +

    +









    + +

    Collect and pre-process data

    @@ -1427,6 +1492,7 @@ and/or if you use anaconda, just write (or install from the graphical use

    # import necessary packages
     import numpy as np
     import matplotlib.pyplot as plt
    +import tensorflow as tf
     from sklearn import datasets
     
     
    @@ -1470,7 +1536,13 @@ plt.show()
     

    -

    from keras.utils import to_categorical
    +
    from tensorflow.keras.layers import Input
    +from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    +from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    +from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    +from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    +from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    +
     from sklearn.model_selection import train_test_split
     
     # one-hot representation of labels
    @@ -1482,266 +1554,10 @@ test_size = 1 - train_size
     X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,
                                                         test_size=test_size)
     
    -

    -









    - -

    Using TensorFlow backend

    - -
      -
    1. Define model and architecture
    2. -
    3. Choose cost function and optimizer
    4. -
    -

    -

    import tensorflow as tf
    -
    -class NeuralNetworkTensorflow:
    -    def __init__(
    -            self,
    -            X_train,
    -            Y_train,
    -            X_test,
    -            Y_test,
    -            n_neurons_layer1=100,
    -            n_neurons_layer2=50,
    -            n_categories=2,
    -            epochs=10,
    -            batch_size=100,
    -            eta=0.1,
    -            lmbd=0.0):
    -        
    -        # keep track of number of steps
    -        self.global_step = tf.Variable(0, dtype=tf.int32, trainable=False, name='global_step')
    -        
    -        self.X_train = X_train
    -        self.Y_train = Y_train
    -        self.X_test = X_test
    -        self.Y_test = Y_test
    -        
    -        self.n_inputs = X_train.shape[0]
    -        self.n_features = X_train.shape[1]
    -        self.n_neurons_layer1 = n_neurons_layer1
    -        self.n_neurons_layer2 = n_neurons_layer2
    -        self.n_categories = n_categories
    -        
    -        self.epochs = epochs
    -        self.batch_size = batch_size
    -        self.iterations = self.n_inputs // self.batch_size
    -        self.eta = eta
    -        self.lmbd = lmbd
    -        
    -        # build network piece by piece
    -        # name scopes (with) are used to enforce creation of new variables
    -        # https://www.tensorflow.org/guide/variables
    -        self.create_placeholders()
    -        self.create_DNN()
    -        self.create_loss()
    -        self.create_optimiser()
    -        self.create_accuracy()
    -    
    -    def create_placeholders(self):
    -        # placeholders are fine here, but "Datasets" are the preferred method
    -        # of streaming data into a model
    -        with tf.name_scope('data'):
    -            self.X = tf.placeholder(tf.float32, shape=(None, self.n_features), name='X_data')
    -            self.Y = tf.placeholder(tf.float32, shape=(None, self.n_categories), name='Y_data')
    -    
    -    def create_DNN(self):
    -        with tf.name_scope('DNN'):
    -            # the weights are stored to calculate regularization loss later
    -            
    -            # Fully connected layer 1
    -            self.W_fc1 = self.weight_variable([self.n_features, self.n_neurons_layer1], name='fc1', dtype=tf.float32)
    -            b_fc1 = self.bias_variable([self.n_neurons_layer1], name='fc1', dtype=tf.float32)
    -            a_fc1 = tf.nn.sigmoid(tf.matmul(self.X, self.W_fc1) + b_fc1)
    -            
    -            # Fully connected layer 2
    -            self.W_fc2 = self.weight_variable([self.n_neurons_layer1, self.n_neurons_layer2], name='fc2', dtype=tf.float32)
    -            b_fc2 = self.bias_variable([self.n_neurons_layer2], name='fc2', dtype=tf.float32)
    -            a_fc2 = tf.nn.sigmoid(tf.matmul(a_fc1, self.W_fc2) + b_fc2)
    -            
    -            # Output layer
    -            self.W_out = self.weight_variable([self.n_neurons_layer2, self.n_categories], name='out', dtype=tf.float32)
    -            b_out = self.bias_variable([self.n_categories], name='out', dtype=tf.float32)
    -            self.z_out = tf.matmul(a_fc2, self.W_out) + b_out
    -    
    -    def create_loss(self):
    -        with tf.name_scope('loss'):
    -            softmax_loss = tf.reduce_mean(tf.nn.softmax_cross_entropy_with_logits_v2(labels=self.Y, logits=self.z_out))
    -            
    -            regularizer_loss_fc1 = tf.nn.l2_loss(self.W_fc1)
    -            regularizer_loss_fc2 = tf.nn.l2_loss(self.W_fc2)
    -            regularizer_loss_out = tf.nn.l2_loss(self.W_out)
    -            regularizer_loss = self.lmbd*(regularizer_loss_fc1 + regularizer_loss_fc2 + regularizer_loss_out)
    -            
    -            self.loss = softmax_loss + regularizer_loss
    -
    -    def create_accuracy(self):
    -        with tf.name_scope('accuracy'):
    -            probabilities = tf.nn.softmax(self.z_out)
    -            predictions = tf.argmax(probabilities, axis=1)
    -            labels = tf.argmax(self.Y, axis=1)
    -            
    -            correct_predictions = tf.equal(predictions, labels)
    -            correct_predictions = tf.cast(correct_predictions, tf.float32)
    -            self.accuracy = tf.reduce_mean(correct_predictions)
    -    
    -    def create_optimiser(self):
    -        with tf.name_scope('optimizer'):
    -            self.optimizer = tf.train.GradientDescentOptimizer(learning_rate=self.eta).minimize(self.loss, global_step=self.global_step)
    -            
    -    def weight_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.truncated_normal(shape, stddev=0.1)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -    
    -    def bias_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.constant(0.1, shape=shape)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -    
    -    def fit(self):
    -        data_indices = np.arange(self.n_inputs)
    -
    -        with tf.Session() as sess:
    -            sess.run(tf.global_variables_initializer())
    -            for i in range(self.epochs):
    -                for j in range(self.iterations):
    -                    chosen_datapoints = np.random.choice(data_indices, size=self.batch_size, replace=False)
    -                    batch_X, batch_Y = self.X_train[chosen_datapoints], self.Y_train[chosen_datapoints]
    -            
    -                    sess.run([DNN.loss, DNN.optimizer],
    -                        feed_dict={DNN.X: batch_X,
    -                                   DNN.Y: batch_Y})
    -                    accuracy = sess.run(DNN.accuracy,
    -                        feed_dict={DNN.X: batch_X,
    -                                   DNN.Y: batch_Y})
    -                    step = sess.run(DNN.global_step)
    -    
    -            self.train_loss, self.train_accuracy = sess.run([DNN.loss, DNN.accuracy],
    -                feed_dict={DNN.X: self.X_train,
    -                           DNN.Y: self.Y_train})
    -        
    -            self.test_loss, self.test_accuracy = sess.run([DNN.loss, DNN.accuracy],
    -                feed_dict={DNN.X: self.X_test,
    -                           DNN.Y: self.Y_test})
    -
    -

    -









    - -

    Optimizing and using gradient descent

    - -

    - - -

    epochs = 100
    -batch_size = 100
    -n_neurons_layer1 = 100
    -n_neurons_layer2 = 50
    -n_categories = 10
    -eta_vals = np.logspace(-5, 1, 7)
    -lmbd_vals = np.logspace(-5, 1, 7)
    -
    -

    - - -

    DNN_tf = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    -        
    -for i, eta in enumerate(eta_vals):
    -    for j, lmbd in enumerate(lmbd_vals):
    -        DNN = NeuralNetworkTensorflow(X_train, Y_train, X_test, Y_test,
    -                                      n_neurons_layer1, n_neurons_layer2, n_categories,
    -                                      epochs=epochs, batch_size=batch_size, eta=eta, lmbd=lmbd)
    -        DNN.fit()
    -        
    -        DNN_tf[i][j] = DNN
    -        
    -        print("Learning rate = ", eta)
    -        print("Lambda = ", lmbd)
    -        print("Test accuracy: %.3f" % DNN.test_accuracy)
    -        print()
    -
    -

    - - -

    # optional
    -# visual representation of grid search
    -# uses seaborn heatmap, could probably do this in matplotlib
    -import seaborn as sns
    -
    -sns.set()
    -
    -train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -
    -for i in range(len(eta_vals)):
    -    for j in range(len(lmbd_vals)):
    -        DNN = DNN_tf[i][j]
    -
    -        train_accuracy[i][j] = DNN.train_accuracy
    -        test_accuracy[i][j] = DNN.test_accuracy
    -
    -        
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Training Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    -
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Test Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    -
    -

    - - -

    # optional
    -# we can use log files to visualize our graph in Tensorboard
    -writer = tf.summary.FileWriter('logs/')
    -writer.add_graph(tf.get_default_graph())
    -
    -

    -









    - -

    Using Keras

    - -

    -Keras is a high level neural network -that supports Tensorflow, CTNK and Theano as backends. -If you have Tensorflow installed Keras is available through the tf.keras module. -If you have Anaconda installed you may run the following command -

    - - -

    conda install keras
    -
    -

    -Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager: - -

    - - -

    pip install keras
    -
    -

    -or look up the instructions here. - -

    - - -

    import tensorflow as tf
    -from tensorflow.keras.layers import Input
    -from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    -from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    -from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    -from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    -from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    -
    -def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):
    +
    def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):
         model = Sequential()
         model.add(Dense(n_neurons_layer1, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
         model.add(Dense(n_neurons_layer2, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
    @@ -1809,7 +1625,7 @@ plt.show()
     











    -

    The Breast Cancer Data, now with Keras

    +

    The Breast Cancer Data, now with Keras

    @@ -1980,6 +1796,52 @@ Test_accuracy=np.zeros((len(n_neuron),'training') plot_data(eta,n_neuron,Test_accuracy, 'testing')

    +

    +









    + +

    Fine-tuning neural network hyperparameters

    + +

    +The flexibility of neural networks is also one of their main +drawbacks: there are many hyperparameters to tweak. Not only can you +use any imaginable network topology (how neurons/nodes are interconnected), +but even in a simple FFNN you can change the number of layers, the +number of neurons per layer, the type of activation function to use in +each layer, the weight initialization logic, the stochastic gradient optmized and much more. How do you +know what combination of hyperparameters is the best for your task? + +

      +
    • You can use grid search with cross-validation to find the right hyperparameters.
    • +
    + +However,since there are many hyperparameters to tune, and since +training a neural network on a large dataset takes a lot of time, you +will only be able to explore a tiny part of the hyperparameter space. + +
      +
    • You can use randomized search.
    • +
    • Or use tools like Oscar, which implements more complex algorithms to help you find a good set of hyperparameters quickly.
    • +
    + +









    + +

    Hidden layers

    + +

    +For many problems you can start with just one or two hidden layers and it will work just fine. +For the MNIST data set you ca easily get a high accuracy using just one hidden layer with a +few hundred neurons. +You can reach for this data set above 98% accuracy using two hidden layers with the same total amount of +neurons, in roughly the same amount of training time. + +

    +For more complex problems, you can gradually +ramp up the number of hidden layers, until you start overfitting the training set. Very complex tasks, such +as large image classification or speech recognition, typically require networks with dozens of layers +and they need a huge amount +of training data. However, you will rarely have to train such networks from scratch: it is much more +common to reuse parts of a pretrained state-of-the-art network that performs a similar task. +

    @@ -2010,9 +1872,27 @@ neural networks suffer from unstable gradients, different layers may learn at widely different speeds

    +









    + +

    More on activation functions, output layers

    + +

    +In most cases you can use the ReLU activation function in the hidden layers (or one of its variants). + +

    +It is a bit faster to compute than other activation functions, and the gradient descent optimization does in general not get stuck. + +

    +For the output layer: + +

      +
    • For classification the softmax activation function is generally a good choice for classification tasks (when the classes are mutually exclusive).
    • +
    • For regression tasks, you can simply use no activation function at all.
    • +
    + -

    Is the Logistic activation function (Sigmoid) our choice?

    +

    Is the Logistic activation function (Sigmoid) our choice?

    Although this unfortunate behavior has been empirically observed for @@ -2042,7 +1922,7 @@ better than the logistic function in deep networks).











    -

    The derivative of the Logistic funtion

    +

    The derivative of the Logistic funtion

    Looking at the logistic activation function, when inputs become large @@ -2078,7 +1958,7 @@ fast to compute).











    -

    The RELU function family

    +

    The RELU function family

    The ReLU activation function suffers from a problem known as the dying @@ -2105,7 +1985,7 @@ $$











    -

    Which activation function should we use?

    +

    Which activation function should we use?

    In general it seems that the ELU activation function is better than @@ -2122,10 +2002,61 @@ want to tweak yet another hyperparameter, you may just use the default spare time and computing power, you can use cross-validation or bootstrap to evaluate other activation functions. +

    +









    + +

    Batch Normalization

    + +

    +Batch Normalization +aims to address the vanishing/exploding gradients problems, and more generally the problem that the +distribution of each layer’s inputs changes during training, as the parameters of the previous layers change. + +

    +The technique consists of adding an operation in the model just before the activation function of each +layer, simply zero-centering and normalizing the inputs, then scaling and shifting the result using two new +parameters per layer (one for scaling, the other for shifting). In other words, this operation lets the model +learn the optimal scale and mean of the inputs for each layer. +In order to zero-center and normalize the inputs, the algorithm needs to estimate the inputs’ mean and +standard deviation. It does so by evaluating the mean and standard deviation of the inputs over the current +mini-batch, from this the name batch normalization. + +

    +









    + +

    Dropout

    + +

    +It is a fairly simple algorithm: at every training step, every neuron (including the input neurons but +excluding the output neurons) has a probability \( p \) of being temporarily dropped out, meaning it will be +entirely ignored during this training step, but it may be active during the next step. + +

    +The +hyperparameter \( p \) is called the dropout rate, and it is typically set to 50%. After training, the neurons are not dropped anymore. + It is viewed as one of the most popular regularization techniques. + +

    +









    + +

    Gradient Clipping

    + +

    +A popular technique to lessen the exploding gradients problem is to simply clip the gradients during +backpropagation so that they never exceed some threshold (this is mostly useful for recurrent neural +networks). + +

    +This technique is called Gradient Clipping. + +

    +In general however, Batch +Normalization is preferred. +

    -

    A top-down perspective on Neural networks

    +

    A top-down perspective on Neural networks

    The first thing we would like to do is divide the data into two or three @@ -2167,7 +2098,7 @@ supervised learning.











    -

    Limitations of supervised learning with deep networks

    +

    Limitations of supervised learning with deep networks

    Like all statistical methods, supervised learning using neural @@ -2193,7 +2124,7 @@ Some of these remarks are particular to DNNs, others are shared by all supervise











    -

    Convolutional Neural Networks (recognizing images)

    +

    Convolutional Neural Networks (recognizing images)

    Convolutional neural networks (CNNs) were developed during the last @@ -2232,7 +2163,7 @@ Another good read is the article here Regular NNs don’t scale well to full images +

    Regular NNs don’t scale well to full images

    As an example, consider @@ -2260,7 +2191,7 @@ would quickly lead to possible overfitting.











    -

    3D volumes of neurons

    +

    3D volumes of neurons

    Convolutional Neural Networks take advantage of the fact that the @@ -2300,7 +2231,7 @@ dimension.

    -

    Layers used to build CNNs

    +

    Layers used to build CNNs

    A simple CNN is a sequence of layers, and every layer of a CNN @@ -2323,7 +2254,7 @@ A simple CNN for image classification could have the architecture:









    -

    Transforming images

    +

    Transforming images

    CNNs transform the original image layer by layer from the original @@ -2342,7 +2273,7 @@ are consistent with the labels in the training set for each image.











    -

    CNNs in brief

    +

    CNNs in brief

    In summary: @@ -2361,6 +2292,291 @@ the course and the slides of CS231 which is taught at Stanford University (consistently ranked as one of the top computer science programs in the world). Michael Nielsen's book is a must read, in particular chapter 6 which deals with CNNs.

    +









    + +

    CNNs in more detail, building convolutional neural networks in Tensorflow and Keras

    + +

    +As discussed above, CNNs are neural networks built from the assumption that the inputs +to the network are 2D images. This is important because the number of features or pixels in images +grows very fast with the image size, and an enormous number of weights and biases are needed in order to build an accurate network. + +

    +As before, we still have our input, a hidden layer and an output. What's novel about convolutional networks +are the convolutional and pooling layers stacked in pairs between the input and the hidden layer. +In addition, the data is no longer represented as a 2D feature matrix, instead each input is a number of 2D +matrices, typically 1 for each color dimension (Red, Green, Blue). + +

    +









    + +

    Setting it up

    + +

    +It means that to represent the entire +dataset of images, we require a 4D matrix or tensor. This tensor has the dimensions: +$$ +(n_{inputs},\, n_{pixels, width},\, n_{pixels, height},\, depth) . +$$ + +

    +









    + +

    The MNIST dataset again

    + +

    +The MNIST dataset consists of grayscale images with a pixel size of +\( 28\times 28 \), meaning we require \( 28 \times 28 = 724 \) weights to each +neuron in the first hidden layer. + +

    +If we were to analyze images of size \( 128\times 128 \) we would require +\( 128 \times 128 = 16384 \) weights to each neuron. Even worse if we were +dealing with color images, as most images are, we have an image matrix +of size \( 128\times 128 \) for each color dimension (Red, Green, Blue), +meaning 3 times the number of weights \( = 49152 \) are required for every +single neuron in the first hidden layer. + +

    +









    + +

    Strong correlations

    + +

    +Images typically have strong local correlations, meaning that a small +part of the image varies little from its neighboring regions. If for +example we have an image of a blue car, we can roughly assume that a +small blue part of the image is surrounded by other blue regions. + +

    +Therefore, instead of connecting every single pixel to a neuron in the +first hidden layer, as we have previously done with deep neural +networks, we can instead connect each neuron to a small part of the +image (in all 3 RGB depth dimensions). The size of each small area is +fixed, and known as a receptive. + +

    + + +

    Layers of a CNN

    +The layers of a convolutional neural network arrange neurons in 3D: width, height and depth. +The input image is typically a square matrix of depth 3. + +

    +A convolution is performed on the image which outputs +a 3D volume of neurons. The weights to the input are arranged in a number of 2D matrices, known as filters. + +

    +Each filter slides along the input image, taking the dot product +between each small part of the image and the filter, in all depth +dimensions. This is then passed through a non-linear function, +typically the Rectified Linear (ReLu) function, which serves as the +activation of the neurons in the first convolutional layer. This is +further passed through a pooling layer, which reduces the size of the +convolutional layer, e.g. by taking the maximum or average across some +small regions, and this serves as input to the next convolutional +layer. + +

    +









    + +

    Systematic reduction

    + +

    +By systematically reducing the size of the input volume, through +convolution and pooling, the network should create representations of +small parts of the input, and then from them assemble representations +of larger areas. The final pooling layer is flattened to serve as +input to a hidden layer, such that each neuron in the final pooling +layer is connected to every single neuron in the hidden layer. This +then serves as input to the output layer, e.g. a softmax output for +classification. + +

    +









    + +

    Prerequisites: Collect and pre-process data

    +

    + + +

    # import necessary packages
    +import numpy as np
    +import matplotlib.pyplot as plt
    +from sklearn import datasets
    +
    +
    +# ensure the same random numbers appear every time
    +np.random.seed(0)
    +
    +# display images in notebook
    +%matplotlib inline
    +plt.rcParams['figure.figsize'] = (12,12)
    +
    +
    +# download MNIST dataset
    +digits = datasets.load_digits()
    +
    +# define inputs and labels
    +inputs = digits.images
    +labels = digits.target
    +
    +# RGB images have a depth of 3
    +# our images are grayscale so they should have a depth of 1
    +inputs = inputs[:,:,:,np.newaxis]
    +
    +print("inputs = (n_inputs, pixel_width, pixel_height, depth) = " + str(inputs.shape))
    +print("labels = (n_inputs) = " + str(labels.shape))
    +
    +
    +# choose some random images to display
    +n_inputs = len(inputs)
    +indices = np.arange(n_inputs)
    +random_indices = np.random.choice(indices, size=5)
    +
    +for i, image in enumerate(digits.images[random_indices]):
    +    plt.subplot(1, 5, i+1)
    +    plt.axis('off')
    +    plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest')
    +    plt.title("Label: %d" % digits.target[random_indices[i]])
    +plt.show()
    +
    +

    +









    + +

    Importing Keras and Tensorflow

    +

    + + +

    from keras.utils import to_categorical
    +from sklearn.model_selection import train_test_split
    +
    +# representation of labels
    +labels = to_categorical(labels)
    +
    +# split into train and test data
    +# one-liner from scikit-learn library
    +train_size = 0.8
    +test_size = 1 - train_size
    +X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,
    +                                                    test_size=test_size)
    +
    +

    + + +

    Running with Keras

    + +

    + + +

    from keras.models import Sequential
    +from keras.layers.convolutional import Conv2D
    +from keras.layers.convolutional import MaxPooling2D
    +from keras.layers import Flatten
    +from keras.layers import Dense
    +from keras.regularizers import l2
    +from keras.optimizers import SGD
    +
    +def create_convolutional_neural_network_keras(input_shape, receptive_field,
    +                                              n_filters, n_neurons_connected, n_categories,
    +                                              eta, lmbd):
    +    model = Sequential()
    +    model.add(Conv2D(n_filters, (receptive_field, receptive_field), input_shape=input_shape, padding='same',
    +              activation='relu', kernel_regularizer=l2(lmbd)))
    +    model.add(MaxPooling2D(pool_size=(2, 2)))
    +    model.add(Flatten())
    +    model.add(Dense(n_neurons_connected, activation='relu', kernel_regularizer=l2(lmbd)))
    +    model.add(Dense(n_categories, activation='softmax', kernel_regularizer=l2(lmbd)))
    +    
    +    sgd = SGD(lr=eta)
    +    model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])
    +    
    +    return model
    +
    +epochs = 100
    +batch_size = 100
    +input_shape = X_train.shape[1:4]
    +receptive_field = 3
    +n_filters = 10
    +n_neurons_connected = 50
    +n_categories = 10
    +
    +eta_vals = np.logspace(-5, 1, 7)
    +lmbd_vals = np.logspace(-5, 1, 7)
    +
    +

    +









    + +

    Final part

    + +

    + + +

    CNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    +        
    +for i, eta in enumerate(eta_vals):
    +    for j, lmbd in enumerate(lmbd_vals):
    +        CNN = create_convolutional_neural_network_keras(input_shape, receptive_field,
    +                                              n_filters, n_neurons_connected, n_categories,
    +                                              eta, lmbd)
    +        CNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0)
    +        scores = CNN.evaluate(X_test, Y_test)
    +        
    +        CNN_keras[i][j] = CNN
    +        
    +        print("Learning rate = ", eta)
    +        print("Lambda = ", lmbd)
    +        print("Test accuracy: %.3f" % scores[1])
    +        print()
    +
    +

    +









    + +

    Final visualization

    + +

    + + +

    # visual representation of grid search
    +# uses seaborn heatmap, could probably do this in matplotlib
    +import seaborn as sns
    +
    +sns.set()
    +
    +train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +
    +for i in range(len(eta_vals)):
    +    for j in range(len(lmbd_vals)):
    +        CNN = CNN_keras[i][j]
    +
    +        train_accuracy[i][j] = CNN.evaluate(X_train, Y_train)[1]
    +        test_accuracy[i][j] = CNN.evaluate(X_test, Y_test)[1]
    +
    +        
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Training Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
    +
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Test Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
    +
    +

    +









    + +

    Fun links

    + +
      +
    1. Self-Driving cars using a convolutional neural network
    2. +
    3. Abstract art using convolutional neural networks
    4. +
    + diff --git a/doc/pub/week41/html/week41.html b/doc/pub/week41/html/week41.html index 5f99792e9..5345112de 100644 --- a/doc/pub/week41/html/week41.html +++ b/doc/pub/week41/html/week41.html @@ -103,39 +103,68 @@ div { text-align: justify; text-justify: inter-word; } None, '___sec25'), ('Tensorflow', 2, None, '___sec26'), - ('Collect and pre-process data', 2, None, '___sec27'), - ('Using TensorFlow backend', 2, None, '___sec28'), - ('Optimizing and using gradient descent', 2, None, '___sec29'), - ('Using Keras', 2, None, '___sec30'), - ('The Breast Cancer Data, now with Keras', 2, None, '___sec31'), + ('Using Keras', 2, None, '___sec27'), + ('Collect and pre-process data', 2, None, '___sec28'), + ('The Breast Cancer Data, now with Keras', 2, None, '___sec29'), + ('Fine-tuning neural network hyperparameters', + 2, + None, + '___sec30'), + ('Hidden layers', 2, None, '___sec31'), ('Which activation function should I use?', 2, None, '___sec32'), - ('Is the Logistic activation function (Sigmoid) our choice?', + ('More on activation functions, output layers', 2, None, '___sec33'), - ('The derivative of the Logistic funtion', 2, None, '___sec34'), - ('The RELU function family', 2, None, '___sec35'), - ('Which activation function should we use?', 2, None, '___sec36'), + ('Is the Logistic activation function (Sigmoid) our choice?', + 2, + None, + '___sec34'), + ('The derivative of the Logistic funtion', 2, None, '___sec35'), + ('The RELU function family', 2, None, '___sec36'), + ('Which activation function should we use?', 2, None, '___sec37'), + ('Batch Normalization', 2, None, '___sec38'), + ('Dropout', 2, None, '___sec39'), + ('Gradient Clipping', 2, None, '___sec40'), ('A top-down perspective on Neural networks', 2, None, - '___sec37'), + '___sec41'), ('Limitations of supervised learning with deep networks', 2, None, - '___sec38'), + '___sec42'), ('Convolutional Neural Networks (recognizing images)', 2, None, - '___sec39'), + '___sec43'), ('Regular NNs don’t scale well to full images', 2, None, - '___sec40'), - ('3D volumes of neurons', 2, None, '___sec41'), - ('Layers used to build CNNs', 2, None, '___sec42'), - ('Transforming images', 2, None, '___sec43'), - ('CNNs in brief', 2, None, '___sec44')]} + '___sec44'), + ('3D volumes of neurons', 2, None, '___sec45'), + ('Layers used to build CNNs', 2, None, '___sec46'), + ('Transforming images', 2, None, '___sec47'), + ('CNNs in brief', 2, None, '___sec48'), + ('CNNs in more detail, building convolutional neural networks in ' + 'Tensorflow and Keras', + 2, + None, + '___sec49'), + ('Setting it up', 2, None, '___sec50'), + ('The MNIST dataset again', 2, None, '___sec51'), + ('Strong correlations', 2, None, '___sec52'), + ('Layers of a CNN', 2, None, '___sec53'), + ('Systematic reduction', 2, None, '___sec54'), + ('Prerequisites: Collect and pre-process data', + 2, + None, + '___sec55'), + ('Importing Keras and Tensorflow', 2, None, '___sec56'), + ('Running with Keras', 2, None, '___sec57'), + ('Final part', 2, None, '___sec58'), + ('Final visualization', 2, None, '___sec59'), + ('Fun links', 2, None, '___sec60')]} end of tocinfo --> @@ -177,7 +206,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 8, 2020

    +

    Oct 9, 2020












    @@ -1416,15 +1445,51 @@ To install tensorflow on Unix/Linux systems, use pip as

    and/or if you use anaconda, just write (or install from the graphical user interface) +(current release of CPU-only TensorFlow)

    -

    conda install tensorflow
    +
    conda create -n tf tensorflow
    +conda activate tf
    +
    +

    +To install the current release of GPU TensorFlow +

    + + +

    conda create -n tf-gpu tensorflow-gpu
    +conda activate tf-gpu
     











    -

    Collect and pre-process data

    +

    Using Keras

    + +

    +Keras is a high level neural network +that supports Tensorflow, CTNK and Theano as backends. +If you have Tensorflow installed Keras is available through the tf.keras module. +If you have Anaconda installed you may run the following command +

    + + +

    conda install keras
    +
    +

    +Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager: + +

    + + +

    pip install keras
    +
    +

    +or look up the instructions here. + +

    +









    + +

    Collect and pre-process data

    @@ -1432,6 +1497,7 @@ and/or if you use anaconda, just write (or install from the graphical use

    # import necessary packages
     import numpy as np
     import matplotlib.pyplot as plt
    +import tensorflow as tf
     from sklearn import datasets
     
     
    @@ -1475,7 +1541,13 @@ plt.show()
     

    -

    from keras.utils import to_categorical
    +
    from tensorflow.keras.layers import Input
    +from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    +from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    +from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    +from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    +from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    +
     from sklearn.model_selection import train_test_split
     
     # one-hot representation of labels
    @@ -1487,266 +1559,10 @@ test_size = 1= train_test_split(inputs, labels, train_size=train_size,
                                                         test_size=test_size)
     
    -

    -









    - -

    Using TensorFlow backend

    - -
      -
    1. Define model and architecture
    2. -
    3. Choose cost function and optimizer
    4. -
    -

    -

    import tensorflow as tf
    -
    -class NeuralNetworkTensorflow:
    -    def __init__(
    -            self,
    -            X_train,
    -            Y_train,
    -            X_test,
    -            Y_test,
    -            n_neurons_layer1=100,
    -            n_neurons_layer2=50,
    -            n_categories=2,
    -            epochs=10,
    -            batch_size=100,
    -            eta=0.1,
    -            lmbd=0.0):
    -        
    -        # keep track of number of steps
    -        self.global_step = tf.Variable(0, dtype=tf.int32, trainable=False, name='global_step')
    -        
    -        self.X_train = X_train
    -        self.Y_train = Y_train
    -        self.X_test = X_test
    -        self.Y_test = Y_test
    -        
    -        self.n_inputs = X_train.shape[0]
    -        self.n_features = X_train.shape[1]
    -        self.n_neurons_layer1 = n_neurons_layer1
    -        self.n_neurons_layer2 = n_neurons_layer2
    -        self.n_categories = n_categories
    -        
    -        self.epochs = epochs
    -        self.batch_size = batch_size
    -        self.iterations = self.n_inputs // self.batch_size
    -        self.eta = eta
    -        self.lmbd = lmbd
    -        
    -        # build network piece by piece
    -        # name scopes (with) are used to enforce creation of new variables
    -        # https://www.tensorflow.org/guide/variables
    -        self.create_placeholders()
    -        self.create_DNN()
    -        self.create_loss()
    -        self.create_optimiser()
    -        self.create_accuracy()
    -    
    -    def create_placeholders(self):
    -        # placeholders are fine here, but "Datasets" are the preferred method
    -        # of streaming data into a model
    -        with tf.name_scope('data'):
    -            self.X = tf.placeholder(tf.float32, shape=(None, self.n_features), name='X_data')
    -            self.Y = tf.placeholder(tf.float32, shape=(None, self.n_categories), name='Y_data')
    -    
    -    def create_DNN(self):
    -        with tf.name_scope('DNN'):
    -            # the weights are stored to calculate regularization loss later
    -            
    -            # Fully connected layer 1
    -            self.W_fc1 = self.weight_variable([self.n_features, self.n_neurons_layer1], name='fc1', dtype=tf.float32)
    -            b_fc1 = self.bias_variable([self.n_neurons_layer1], name='fc1', dtype=tf.float32)
    -            a_fc1 = tf.nn.sigmoid(tf.matmul(self.X, self.W_fc1) + b_fc1)
    -            
    -            # Fully connected layer 2
    -            self.W_fc2 = self.weight_variable([self.n_neurons_layer1, self.n_neurons_layer2], name='fc2', dtype=tf.float32)
    -            b_fc2 = self.bias_variable([self.n_neurons_layer2], name='fc2', dtype=tf.float32)
    -            a_fc2 = tf.nn.sigmoid(tf.matmul(a_fc1, self.W_fc2) + b_fc2)
    -            
    -            # Output layer
    -            self.W_out = self.weight_variable([self.n_neurons_layer2, self.n_categories], name='out', dtype=tf.float32)
    -            b_out = self.bias_variable([self.n_categories], name='out', dtype=tf.float32)
    -            self.z_out = tf.matmul(a_fc2, self.W_out) + b_out
    -    
    -    def create_loss(self):
    -        with tf.name_scope('loss'):
    -            softmax_loss = tf.reduce_mean(tf.nn.softmax_cross_entropy_with_logits_v2(labels=self.Y, logits=self.z_out))
    -            
    -            regularizer_loss_fc1 = tf.nn.l2_loss(self.W_fc1)
    -            regularizer_loss_fc2 = tf.nn.l2_loss(self.W_fc2)
    -            regularizer_loss_out = tf.nn.l2_loss(self.W_out)
    -            regularizer_loss = self.lmbd*(regularizer_loss_fc1 + regularizer_loss_fc2 + regularizer_loss_out)
    -            
    -            self.loss = softmax_loss + regularizer_loss
    -
    -    def create_accuracy(self):
    -        with tf.name_scope('accuracy'):
    -            probabilities = tf.nn.softmax(self.z_out)
    -            predictions = tf.argmax(probabilities, axis=1)
    -            labels = tf.argmax(self.Y, axis=1)
    -            
    -            correct_predictions = tf.equal(predictions, labels)
    -            correct_predictions = tf.cast(correct_predictions, tf.float32)
    -            self.accuracy = tf.reduce_mean(correct_predictions)
    -    
    -    def create_optimiser(self):
    -        with tf.name_scope('optimizer'):
    -            self.optimizer = tf.train.GradientDescentOptimizer(learning_rate=self.eta).minimize(self.loss, global_step=self.global_step)
    -            
    -    def weight_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.truncated_normal(shape, stddev=0.1)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -    
    -    def bias_variable(self, shape, name='', dtype=tf.float32):
    -        initial = tf.constant(0.1, shape=shape)
    -        return tf.Variable(initial, name=name, dtype=dtype)
    -    
    -    def fit(self):
    -        data_indices = np.arange(self.n_inputs)
    -
    -        with tf.Session() as sess:
    -            sess.run(tf.global_variables_initializer())
    -            for i in range(self.epochs):
    -                for j in range(self.iterations):
    -                    chosen_datapoints = np.random.choice(data_indices, size=self.batch_size, replace=False)
    -                    batch_X, batch_Y = self.X_train[chosen_datapoints], self.Y_train[chosen_datapoints]
    -            
    -                    sess.run([DNN.loss, DNN.optimizer],
    -                        feed_dict={DNN.X: batch_X,
    -                                   DNN.Y: batch_Y})
    -                    accuracy = sess.run(DNN.accuracy,
    -                        feed_dict={DNN.X: batch_X,
    -                                   DNN.Y: batch_Y})
    -                    step = sess.run(DNN.global_step)
    -    
    -            self.train_loss, self.train_accuracy = sess.run([DNN.loss, DNN.accuracy],
    -                feed_dict={DNN.X: self.X_train,
    -                           DNN.Y: self.Y_train})
    -        
    -            self.test_loss, self.test_accuracy = sess.run([DNN.loss, DNN.accuracy],
    -                feed_dict={DNN.X: self.X_test,
    -                           DNN.Y: self.Y_test})
    -
    -

    -









    - -

    Optimizing and using gradient descent

    - -

    - - -

    epochs = 100
    -batch_size = 100
    -n_neurons_layer1 = 100
    -n_neurons_layer2 = 50
    -n_categories = 10
    -eta_vals = np.logspace(-5, 1, 7)
    -lmbd_vals = np.logspace(-5, 1, 7)
    -
    -

    - - -

    DNN_tf = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    -        
    -for i, eta in enumerate(eta_vals):
    -    for j, lmbd in enumerate(lmbd_vals):
    -        DNN = NeuralNetworkTensorflow(X_train, Y_train, X_test, Y_test,
    -                                      n_neurons_layer1, n_neurons_layer2, n_categories,
    -                                      epochs=epochs, batch_size=batch_size, eta=eta, lmbd=lmbd)
    -        DNN.fit()
    -        
    -        DNN_tf[i][j] = DNN
    -        
    -        print("Learning rate = ", eta)
    -        print("Lambda = ", lmbd)
    -        print("Test accuracy: %.3f" % DNN.test_accuracy)
    -        print()
    -
    -

    - - -

    # optional
    -# visual representation of grid search
    -# uses seaborn heatmap, could probably do this in matplotlib
    -import seaborn as sns
    -
    -sns.set()
    -
    -train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    -
    -for i in range(len(eta_vals)):
    -    for j in range(len(lmbd_vals)):
    -        DNN = DNN_tf[i][j]
    -
    -        train_accuracy[i][j] = DNN.train_accuracy
    -        test_accuracy[i][j] = DNN.test_accuracy
    -
    -        
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Training Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    -
    -fig, ax = plt.subplots(figsize = (10, 10))
    -sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    -ax.set_title("Test Accuracy")
    -ax.set_ylabel("$\eta$")
    -ax.set_xlabel("$\lambda$")
    -plt.show()
    -
    -

    - - -

    # optional
    -# we can use log files to visualize our graph in Tensorboard
    -writer = tf.summary.FileWriter('logs/')
    -writer.add_graph(tf.get_default_graph())
    -
    -

    -









    - -

    Using Keras

    - -

    -Keras is a high level neural network -that supports Tensorflow, CTNK and Theano as backends. -If you have Tensorflow installed Keras is available through the tf.keras module. -If you have Anaconda installed you may run the following command -

    - - -

    conda install keras
    -
    -

    -Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager: - -

    - - -

    pip install keras
    -
    -

    -or look up the instructions here. - -

    - - -

    import tensorflow as tf
    -from tensorflow.keras.layers import Input
    -from tensorflow.keras.models import Sequential      #This allows appending layers to existing models
    -from tensorflow.keras.layers import Dense           #This allows defining the characteristics of a particular layer
    -from tensorflow.keras import optimizers             #This allows using whichever optimiser we want (sgd,adam,RMSprop)
    -from tensorflow.keras import regularizers           #This allows using whichever regularizer we want (l1,l2,l1_l2)
    -from tensorflow.keras.utils import to_categorical   #This allows using categorical cross entropy as the cost function
    -
    -def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):
    +
    def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):
         model = Sequential()
         model.add(Dense(n_neurons_layer1, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
         model.add(Dense(n_neurons_layer2, activation='sigmoid', kernel_regularizer=regularizers.l2(lmbd)))
    @@ -1814,7 +1630,7 @@ plt.show()
     











    -

    The Breast Cancer Data, now with Keras

    +

    The Breast Cancer Data, now with Keras

    @@ -1985,6 +1801,52 @@ Test_accuracy=np'training') plot_data(eta,n_neuron,Test_accuracy, 'testing')

    +

    +









    + +

    Fine-tuning neural network hyperparameters

    + +

    +The flexibility of neural networks is also one of their main +drawbacks: there are many hyperparameters to tweak. Not only can you +use any imaginable network topology (how neurons/nodes are interconnected), +but even in a simple FFNN you can change the number of layers, the +number of neurons per layer, the type of activation function to use in +each layer, the weight initialization logic, the stochastic gradient optmized and much more. How do you +know what combination of hyperparameters is the best for your task? + +

      +
    • You can use grid search with cross-validation to find the right hyperparameters.
    • +
    + +However,since there are many hyperparameters to tune, and since +training a neural network on a large dataset takes a lot of time, you +will only be able to explore a tiny part of the hyperparameter space. + +
      +
    • You can use randomized search.
    • +
    • Or use tools like Oscar, which implements more complex algorithms to help you find a good set of hyperparameters quickly.
    • +
    + +









    + +

    Hidden layers

    + +

    +For many problems you can start with just one or two hidden layers and it will work just fine. +For the MNIST data set you ca easily get a high accuracy using just one hidden layer with a +few hundred neurons. +You can reach for this data set above 98% accuracy using two hidden layers with the same total amount of +neurons, in roughly the same amount of training time. + +

    +For more complex problems, you can gradually +ramp up the number of hidden layers, until you start overfitting the training set. Very complex tasks, such +as large image classification or speech recognition, typically require networks with dozens of layers +and they need a huge amount +of training data. However, you will rarely have to train such networks from scratch: it is much more +common to reuse parts of a pretrained state-of-the-art network that performs a similar task. +

    @@ -2015,9 +1877,27 @@ neural networks suffer from unstable gradients, different layers may learn at widely different speeds

    +









    + +

    More on activation functions, output layers

    + +

    +In most cases you can use the ReLU activation function in the hidden layers (or one of its variants). + +

    +It is a bit faster to compute than other activation functions, and the gradient descent optimization does in general not get stuck. + +

    +For the output layer: + +

      +
    • For classification the softmax activation function is generally a good choice for classification tasks (when the classes are mutually exclusive).
    • +
    • For regression tasks, you can simply use no activation function at all.
    • +
    + -

    Is the Logistic activation function (Sigmoid) our choice?

    +

    Is the Logistic activation function (Sigmoid) our choice?

    Although this unfortunate behavior has been empirically observed for @@ -2047,7 +1927,7 @@ better than the logistic function in deep networks).











    -

    The derivative of the Logistic funtion

    +

    The derivative of the Logistic funtion

    Looking at the logistic activation function, when inputs become large @@ -2083,7 +1963,7 @@ fast to compute).











    -

    The RELU function family

    +

    The RELU function family

    The ReLU activation function suffers from a problem known as the dying @@ -2110,7 +1990,7 @@ $$











    -

    Which activation function should we use?

    +

    Which activation function should we use?

    In general it seems that the ELU activation function is better than @@ -2127,10 +2007,61 @@ want to tweak yet another hyperparameter, you may just use the default spare time and computing power, you can use cross-validation or bootstrap to evaluate other activation functions. +

    +









    + +

    Batch Normalization

    + +

    +Batch Normalization +aims to address the vanishing/exploding gradients problems, and more generally the problem that the +distribution of each layer’s inputs changes during training, as the parameters of the previous layers change. + +

    +The technique consists of adding an operation in the model just before the activation function of each +layer, simply zero-centering and normalizing the inputs, then scaling and shifting the result using two new +parameters per layer (one for scaling, the other for shifting). In other words, this operation lets the model +learn the optimal scale and mean of the inputs for each layer. +In order to zero-center and normalize the inputs, the algorithm needs to estimate the inputs’ mean and +standard deviation. It does so by evaluating the mean and standard deviation of the inputs over the current +mini-batch, from this the name batch normalization. + +

    +









    + +

    Dropout

    + +

    +It is a fairly simple algorithm: at every training step, every neuron (including the input neurons but +excluding the output neurons) has a probability \( p \) of being temporarily dropped out, meaning it will be +entirely ignored during this training step, but it may be active during the next step. + +

    +The +hyperparameter \( p \) is called the dropout rate, and it is typically set to 50%. After training, the neurons are not dropped anymore. + It is viewed as one of the most popular regularization techniques. + +

    +









    + +

    Gradient Clipping

    + +

    +A popular technique to lessen the exploding gradients problem is to simply clip the gradients during +backpropagation so that they never exceed some threshold (this is mostly useful for recurrent neural +networks). + +

    +This technique is called Gradient Clipping. + +

    +In general however, Batch +Normalization is preferred. +

    -

    A top-down perspective on Neural networks

    +

    A top-down perspective on Neural networks

    The first thing we would like to do is divide the data into two or three @@ -2172,7 +2103,7 @@ supervised learning.











    -

    Limitations of supervised learning with deep networks

    +

    Limitations of supervised learning with deep networks

    Like all statistical methods, supervised learning using neural @@ -2198,7 +2129,7 @@ Some of these remarks are particular to DNNs, others are shared by all supervise











    -

    Convolutional Neural Networks (recognizing images)

    +

    Convolutional Neural Networks (recognizing images)

    Convolutional neural networks (CNNs) were developed during the last @@ -2237,7 +2168,7 @@ Another good read is the article here Regular NNs don’t scale well to full images +

    Regular NNs don’t scale well to full images

    As an example, consider @@ -2265,7 +2196,7 @@ would quickly lead to possible overfitting.











    -

    3D volumes of neurons

    +

    3D volumes of neurons

    Convolutional Neural Networks take advantage of the fact that the @@ -2305,7 +2236,7 @@ dimension.

    -

    Layers used to build CNNs

    +

    Layers used to build CNNs

    A simple CNN is a sequence of layers, and every layer of a CNN @@ -2328,7 +2259,7 @@ A simple CNN for image classification could have the architecture:









    -

    Transforming images

    +

    Transforming images

    CNNs transform the original image layer by layer from the original @@ -2347,7 +2278,7 @@ are consistent with the labels in the training set for each image.











    -

    CNNs in brief

    +

    CNNs in brief

    In summary: @@ -2366,6 +2297,291 @@ the course and the slides of CS231 which is taught at Stanford University (consistently ranked as one of the top computer science programs in the world). Michael Nielsen's book is a must read, in particular chapter 6 which deals with CNNs.

    +









    + +

    CNNs in more detail, building convolutional neural networks in Tensorflow and Keras

    + +

    +As discussed above, CNNs are neural networks built from the assumption that the inputs +to the network are 2D images. This is important because the number of features or pixels in images +grows very fast with the image size, and an enormous number of weights and biases are needed in order to build an accurate network. + +

    +As before, we still have our input, a hidden layer and an output. What's novel about convolutional networks +are the convolutional and pooling layers stacked in pairs between the input and the hidden layer. +In addition, the data is no longer represented as a 2D feature matrix, instead each input is a number of 2D +matrices, typically 1 for each color dimension (Red, Green, Blue). + +

    +









    + +

    Setting it up

    + +

    +It means that to represent the entire +dataset of images, we require a 4D matrix or tensor. This tensor has the dimensions: +$$ +(n_{inputs},\, n_{pixels, width},\, n_{pixels, height},\, depth) . +$$ + +

    +









    + +

    The MNIST dataset again

    + +

    +The MNIST dataset consists of grayscale images with a pixel size of +\( 28\times 28 \), meaning we require \( 28 \times 28 = 724 \) weights to each +neuron in the first hidden layer. + +

    +If we were to analyze images of size \( 128\times 128 \) we would require +\( 128 \times 128 = 16384 \) weights to each neuron. Even worse if we were +dealing with color images, as most images are, we have an image matrix +of size \( 128\times 128 \) for each color dimension (Red, Green, Blue), +meaning 3 times the number of weights \( = 49152 \) are required for every +single neuron in the first hidden layer. + +

    +









    + +

    Strong correlations

    + +

    +Images typically have strong local correlations, meaning that a small +part of the image varies little from its neighboring regions. If for +example we have an image of a blue car, we can roughly assume that a +small blue part of the image is surrounded by other blue regions. + +

    +Therefore, instead of connecting every single pixel to a neuron in the +first hidden layer, as we have previously done with deep neural +networks, we can instead connect each neuron to a small part of the +image (in all 3 RGB depth dimensions). The size of each small area is +fixed, and known as a receptive. + +

    + + +

    Layers of a CNN

    +The layers of a convolutional neural network arrange neurons in 3D: width, height and depth. +The input image is typically a square matrix of depth 3. + +

    +A convolution is performed on the image which outputs +a 3D volume of neurons. The weights to the input are arranged in a number of 2D matrices, known as filters. + +

    +Each filter slides along the input image, taking the dot product +between each small part of the image and the filter, in all depth +dimensions. This is then passed through a non-linear function, +typically the Rectified Linear (ReLu) function, which serves as the +activation of the neurons in the first convolutional layer. This is +further passed through a pooling layer, which reduces the size of the +convolutional layer, e.g. by taking the maximum or average across some +small regions, and this serves as input to the next convolutional +layer. + +

    +









    + +

    Systematic reduction

    + +

    +By systematically reducing the size of the input volume, through +convolution and pooling, the network should create representations of +small parts of the input, and then from them assemble representations +of larger areas. The final pooling layer is flattened to serve as +input to a hidden layer, such that each neuron in the final pooling +layer is connected to every single neuron in the hidden layer. This +then serves as input to the output layer, e.g. a softmax output for +classification. + +

    +









    + +

    Prerequisites: Collect and pre-process data

    +

    + + +

    # import necessary packages
    +import numpy as np
    +import matplotlib.pyplot as plt
    +from sklearn import datasets
    +
    +
    +# ensure the same random numbers appear every time
    +np.random.seed(0)
    +
    +# display images in notebook
    +%matplotlib inline
    +plt.rcParams['figure.figsize'] = (12,12)
    +
    +
    +# download MNIST dataset
    +digits = datasets.load_digits()
    +
    +# define inputs and labels
    +inputs = digits.images
    +labels = digits.target
    +
    +# RGB images have a depth of 3
    +# our images are grayscale so they should have a depth of 1
    +inputs = inputs[:,:,:,np.newaxis]
    +
    +print("inputs = (n_inputs, pixel_width, pixel_height, depth) = " + str(inputs.shape))
    +print("labels = (n_inputs) = " + str(labels.shape))
    +
    +
    +# choose some random images to display
    +n_inputs = len(inputs)
    +indices = np.arange(n_inputs)
    +random_indices = np.random.choice(indices, size=5)
    +
    +for i, image in enumerate(digits.images[random_indices]):
    +    plt.subplot(1, 5, i+1)
    +    plt.axis('off')
    +    plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest')
    +    plt.title("Label: %d" % digits.target[random_indices[i]])
    +plt.show()
    +
    +

    +









    + +

    Importing Keras and Tensorflow

    +

    + + +

    from keras.utils import to_categorical
    +from sklearn.model_selection import train_test_split
    +
    +# representation of labels
    +labels = to_categorical(labels)
    +
    +# split into train and test data
    +# one-liner from scikit-learn library
    +train_size = 0.8
    +test_size = 1 - train_size
    +X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,
    +                                                    test_size=test_size)
    +
    +

    + + +

    Running with Keras

    + +

    + + +

    from keras.models import Sequential
    +from keras.layers.convolutional import Conv2D
    +from keras.layers.convolutional import MaxPooling2D
    +from keras.layers import Flatten
    +from keras.layers import Dense
    +from keras.regularizers import l2
    +from keras.optimizers import SGD
    +
    +def create_convolutional_neural_network_keras(input_shape, receptive_field,
    +                                              n_filters, n_neurons_connected, n_categories,
    +                                              eta, lmbd):
    +    model = Sequential()
    +    model.add(Conv2D(n_filters, (receptive_field, receptive_field), input_shape=input_shape, padding='same',
    +              activation='relu', kernel_regularizer=l2(lmbd)))
    +    model.add(MaxPooling2D(pool_size=(2, 2)))
    +    model.add(Flatten())
    +    model.add(Dense(n_neurons_connected, activation='relu', kernel_regularizer=l2(lmbd)))
    +    model.add(Dense(n_categories, activation='softmax', kernel_regularizer=l2(lmbd)))
    +    
    +    sgd = SGD(lr=eta)
    +    model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])
    +    
    +    return model
    +
    +epochs = 100
    +batch_size = 100
    +input_shape = X_train.shape[1:4]
    +receptive_field = 3
    +n_filters = 10
    +n_neurons_connected = 50
    +n_categories = 10
    +
    +eta_vals = np.logspace(-5, 1, 7)
    +lmbd_vals = np.logspace(-5, 1, 7)
    +
    +

    +









    + +

    Final part

    + +

    + + +

    CNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
    +        
    +for i, eta in enumerate(eta_vals):
    +    for j, lmbd in enumerate(lmbd_vals):
    +        CNN = create_convolutional_neural_network_keras(input_shape, receptive_field,
    +                                              n_filters, n_neurons_connected, n_categories,
    +                                              eta, lmbd)
    +        CNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0)
    +        scores = CNN.evaluate(X_test, Y_test)
    +        
    +        CNN_keras[i][j] = CNN
    +        
    +        print("Learning rate = ", eta)
    +        print("Lambda = ", lmbd)
    +        print("Test accuracy: %.3f" % scores[1])
    +        print()
    +
    +

    +









    + +

    Final visualization

    + +

    + + +

    # visual representation of grid search
    +# uses seaborn heatmap, could probably do this in matplotlib
    +import seaborn as sns
    +
    +sns.set()
    +
    +train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
    +
    +for i in range(len(eta_vals)):
    +    for j in range(len(lmbd_vals)):
    +        CNN = CNN_keras[i][j]
    +
    +        train_accuracy[i][j] = CNN.evaluate(X_train, Y_train)[1]
    +        test_accuracy[i][j] = CNN.evaluate(X_test, Y_test)[1]
    +
    +        
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Training Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
    +
    +fig, ax = plt.subplots(figsize = (10, 10))
    +sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis")
    +ax.set_title("Test Accuracy")
    +ax.set_ylabel("$\eta$")
    +ax.set_xlabel("$\lambda$")
    +plt.show()
    +
    +

    +









    + +

    Fun links

    + +
      +
    1. Self-Driving cars using a convolutional neural network
    2. +
    3. Abstract art using convolutional neural networks
    4. +
    + diff --git a/doc/pub/week41/ipynb/ipynb-week41-src.tar.gz b/doc/pub/week41/ipynb/ipynb-week41-src.tar.gz index 0833502bbf8900144451eb2294eca3b22d1e9549..b7f9e9bdee0932e9af6c875e48e4e17abd0d12d9 100644 GIT binary patch delta 20 bcmdn6igm*(RyO%=4hFl2jci-l7_~wHOIQY2 delta 20 bcmdn6igm*(RyO%=4u \n", "**Morten Hjorth-Jensen**, Department of Physics, University of Oslo and Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University\n", "\n", - "Date: **Oct 8, 2020**\n", + "Date: **Oct 9, 2020**\n", "\n", "Copyright 1999-2020, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license\n", "\n", @@ -1422,7 +1422,8 @@ "cell_type": "markdown", "metadata": {}, "source": [ - "and/or if you use **anaconda**, just write (or install from the graphical user interface)" + "and/or if you use **anaconda**, just write (or install from the graphical user interface)\n", + "(current release of CPU-only TensorFlow)" ] }, { @@ -1433,14 +1434,15 @@ }, "outputs": [], "source": [ - "conda install tensorflow" + "conda create -n tf tensorflow\n", + "conda activate tf" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ - "## Collect and pre-process data" + "To install the current release of GPU TensorFlow" ] }, { @@ -1450,10 +1452,74 @@ "collapsed": false }, "outputs": [], + "source": [ + "conda create -n tf-gpu tensorflow-gpu\n", + "conda activate tf-gpu" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Using Keras\n", + "\n", + "Keras is a high level [neural network](https://en.wikipedia.org/wiki/Application_programming_interface)\n", + "that supports Tensorflow, CTNK and Theano as backends. \n", + "If you have Tensorflow installed Keras is available through the *tf.keras* module. \n", + "If you have Anaconda installed you may run the following command" + ] + }, + { + "cell_type": "code", + "execution_count": 15, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "conda install keras" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager:" + ] + }, + { + "cell_type": "code", + "execution_count": 16, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "pip install keras" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or look up the [instructions here](https://keras.io/).\n", + "\n", + "\n", + "## Collect and pre-process data" + ] + }, + { + "cell_type": "code", + "execution_count": 17, + "metadata": { + "collapsed": false + }, + "outputs": [], "source": [ "# import necessary packages\n", "import numpy as np\n", "import matplotlib.pyplot as plt\n", + "import tensorflow as tf\n", "from sklearn import datasets\n", "\n", "\n", @@ -1497,13 +1563,19 @@ }, { "cell_type": "code", - "execution_count": 15, + "execution_count": 18, "metadata": { "collapsed": false }, "outputs": [], "source": [ - "from keras.utils import to_categorical\n", + "from tensorflow.keras.layers import Input\n", + "from tensorflow.keras.models import Sequential #This allows appending layers to existing models\n", + "from tensorflow.keras.layers import Dense #This allows defining the characteristics of a particular layer\n", + "from tensorflow.keras import optimizers #This allows using whichever optimiser we want (sgd,adam,RMSprop)\n", + "from tensorflow.keras import regularizers #This allows using whichever regularizer we want (l1,l2,l1_l2)\n", + "from tensorflow.keras.utils import to_categorical #This allows using categorical cross entropy as the cost function\n", + "\n", "from sklearn.model_selection import train_test_split\n", "\n", "# one-hot representation of labels\n", @@ -1516,206 +1588,6 @@ " test_size=test_size)" ] }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Using TensorFlow backend\n", - "\n", - "1. Define model and architecture\n", - "\n", - "2. Choose cost function and optimizer" - ] - }, - { - "cell_type": "code", - "execution_count": 16, - "metadata": { - "collapsed": false - }, - "outputs": [], - "source": [ - "import tensorflow as tf\n", - "\n", - "class NeuralNetworkTensorflow:\n", - " def __init__(\n", - " self,\n", - " X_train,\n", - " Y_train,\n", - " X_test,\n", - " Y_test,\n", - " n_neurons_layer1=100,\n", - " n_neurons_layer2=50,\n", - " n_categories=2,\n", - " epochs=10,\n", - " batch_size=100,\n", - " eta=0.1,\n", - " lmbd=0.0):\n", - " \n", - " # keep track of number of steps\n", - " self.global_step = tf.Variable(0, dtype=tf.int32, trainable=False, name='global_step')\n", - " \n", - " self.X_train = X_train\n", - " self.Y_train = Y_train\n", - " self.X_test = X_test\n", - " self.Y_test = Y_test\n", - " \n", - " self.n_inputs = X_train.shape[0]\n", - " self.n_features = X_train.shape[1]\n", - " self.n_neurons_layer1 = n_neurons_layer1\n", - " self.n_neurons_layer2 = n_neurons_layer2\n", - " self.n_categories = n_categories\n", - " \n", - " self.epochs = epochs\n", - " self.batch_size = batch_size\n", - " self.iterations = self.n_inputs // self.batch_size\n", - " self.eta = eta\n", - " self.lmbd = lmbd\n", - " \n", - " # build network piece by piece\n", - " # name scopes (with) are used to enforce creation of new variables\n", - " # https://www.tensorflow.org/guide/variables\n", - " self.create_placeholders()\n", - " self.create_DNN()\n", - " self.create_loss()\n", - " self.create_optimiser()\n", - " self.create_accuracy()\n", - " \n", - " def create_placeholders(self):\n", - " # placeholders are fine here, but \"Datasets\" are the preferred method\n", - " # of streaming data into a model\n", - " with tf.name_scope('data'):\n", - " self.X = tf.placeholder(tf.float32, shape=(None, self.n_features), name='X_data')\n", - " self.Y = tf.placeholder(tf.float32, shape=(None, self.n_categories), name='Y_data')\n", - " \n", - " def create_DNN(self):\n", - " with tf.name_scope('DNN'):\n", - " # the weights are stored to calculate regularization loss later\n", - " \n", - " # Fully connected layer 1\n", - " self.W_fc1 = self.weight_variable([self.n_features, self.n_neurons_layer1], name='fc1', dtype=tf.float32)\n", - " b_fc1 = self.bias_variable([self.n_neurons_layer1], name='fc1', dtype=tf.float32)\n", - " a_fc1 = tf.nn.sigmoid(tf.matmul(self.X, self.W_fc1) + b_fc1)\n", - " \n", - " # Fully connected layer 2\n", - " self.W_fc2 = self.weight_variable([self.n_neurons_layer1, self.n_neurons_layer2], name='fc2', dtype=tf.float32)\n", - " b_fc2 = self.bias_variable([self.n_neurons_layer2], name='fc2', dtype=tf.float32)\n", - " a_fc2 = tf.nn.sigmoid(tf.matmul(a_fc1, self.W_fc2) + b_fc2)\n", - " \n", - " # Output layer\n", - " self.W_out = self.weight_variable([self.n_neurons_layer2, self.n_categories], name='out', dtype=tf.float32)\n", - " b_out = self.bias_variable([self.n_categories], name='out', dtype=tf.float32)\n", - " self.z_out = tf.matmul(a_fc2, self.W_out) + b_out\n", - " \n", - " def create_loss(self):\n", - " with tf.name_scope('loss'):\n", - " softmax_loss = tf.reduce_mean(tf.nn.softmax_cross_entropy_with_logits_v2(labels=self.Y, logits=self.z_out))\n", - " \n", - " regularizer_loss_fc1 = tf.nn.l2_loss(self.W_fc1)\n", - " regularizer_loss_fc2 = tf.nn.l2_loss(self.W_fc2)\n", - " regularizer_loss_out = tf.nn.l2_loss(self.W_out)\n", - " regularizer_loss = self.lmbd*(regularizer_loss_fc1 + regularizer_loss_fc2 + regularizer_loss_out)\n", - " \n", - " self.loss = softmax_loss + regularizer_loss\n", - "\n", - " def create_accuracy(self):\n", - " with tf.name_scope('accuracy'):\n", - " probabilities = tf.nn.softmax(self.z_out)\n", - " predictions = tf.argmax(probabilities, axis=1)\n", - " labels = tf.argmax(self.Y, axis=1)\n", - " \n", - " correct_predictions = tf.equal(predictions, labels)\n", - " correct_predictions = tf.cast(correct_predictions, tf.float32)\n", - " self.accuracy = tf.reduce_mean(correct_predictions)\n", - " \n", - " def create_optimiser(self):\n", - " with tf.name_scope('optimizer'):\n", - " self.optimizer = tf.train.GradientDescentOptimizer(learning_rate=self.eta).minimize(self.loss, global_step=self.global_step)\n", - " \n", - " def weight_variable(self, shape, name='', dtype=tf.float32):\n", - " initial = tf.truncated_normal(shape, stddev=0.1)\n", - " return tf.Variable(initial, name=name, dtype=dtype)\n", - " \n", - " def bias_variable(self, shape, name='', dtype=tf.float32):\n", - " initial = tf.constant(0.1, shape=shape)\n", - " return tf.Variable(initial, name=name, dtype=dtype)\n", - " \n", - " def fit(self):\n", - " data_indices = np.arange(self.n_inputs)\n", - "\n", - " with tf.Session() as sess:\n", - " sess.run(tf.global_variables_initializer())\n", - " for i in range(self.epochs):\n", - " for j in range(self.iterations):\n", - " chosen_datapoints = np.random.choice(data_indices, size=self.batch_size, replace=False)\n", - " batch_X, batch_Y = self.X_train[chosen_datapoints], self.Y_train[chosen_datapoints]\n", - " \n", - " sess.run([DNN.loss, DNN.optimizer],\n", - " feed_dict={DNN.X: batch_X,\n", - " DNN.Y: batch_Y})\n", - " accuracy = sess.run(DNN.accuracy,\n", - " feed_dict={DNN.X: batch_X,\n", - " DNN.Y: batch_Y})\n", - " step = sess.run(DNN.global_step)\n", - " \n", - " self.train_loss, self.train_accuracy = sess.run([DNN.loss, DNN.accuracy],\n", - " feed_dict={DNN.X: self.X_train,\n", - " DNN.Y: self.Y_train})\n", - " \n", - " self.test_loss, self.test_accuracy = sess.run([DNN.loss, DNN.accuracy],\n", - " feed_dict={DNN.X: self.X_test,\n", - " DNN.Y: self.Y_test})" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Optimizing and using gradient descent" - ] - }, - { - "cell_type": "code", - "execution_count": 17, - "metadata": { - "collapsed": false - }, - "outputs": [], - "source": [ - "epochs = 100\n", - "batch_size = 100\n", - "n_neurons_layer1 = 100\n", - "n_neurons_layer2 = 50\n", - "n_categories = 10\n", - "eta_vals = np.logspace(-5, 1, 7)\n", - "lmbd_vals = np.logspace(-5, 1, 7)" - ] - }, - { - "cell_type": "code", - "execution_count": 18, - "metadata": { - "collapsed": false - }, - "outputs": [], - "source": [ - "DNN_tf = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)\n", - " \n", - "for i, eta in enumerate(eta_vals):\n", - " for j, lmbd in enumerate(lmbd_vals):\n", - " DNN = NeuralNetworkTensorflow(X_train, Y_train, X_test, Y_test,\n", - " n_neurons_layer1, n_neurons_layer2, n_categories,\n", - " epochs=epochs, batch_size=batch_size, eta=eta, lmbd=lmbd)\n", - " DNN.fit()\n", - " \n", - " DNN_tf[i][j] = DNN\n", - " \n", - " print(\"Learning rate = \", eta)\n", - " print(\"Lambda = \", lmbd)\n", - " print(\"Test accuracy: %.3f\" % DNN.test_accuracy)\n", - " print()" - ] - }, { "cell_type": "code", "execution_count": 19, @@ -1724,116 +1596,6 @@ }, "outputs": [], "source": [ - "# optional\n", - "# visual representation of grid search\n", - "# uses seaborn heatmap, could probably do this in matplotlib\n", - "import seaborn as sns\n", - "\n", - "sns.set()\n", - "\n", - "train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))\n", - "test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))\n", - "\n", - "for i in range(len(eta_vals)):\n", - " for j in range(len(lmbd_vals)):\n", - " DNN = DNN_tf[i][j]\n", - "\n", - " train_accuracy[i][j] = DNN.train_accuracy\n", - " test_accuracy[i][j] = DNN.test_accuracy\n", - "\n", - " \n", - "fig, ax = plt.subplots(figsize = (10, 10))\n", - "sns.heatmap(train_accuracy, annot=True, ax=ax, cmap=\"viridis\")\n", - "ax.set_title(\"Training Accuracy\")\n", - "ax.set_ylabel(\"$\\eta$\")\n", - "ax.set_xlabel(\"$\\lambda$\")\n", - "plt.show()\n", - "\n", - "fig, ax = plt.subplots(figsize = (10, 10))\n", - "sns.heatmap(test_accuracy, annot=True, ax=ax, cmap=\"viridis\")\n", - "ax.set_title(\"Test Accuracy\")\n", - "ax.set_ylabel(\"$\\eta$\")\n", - "ax.set_xlabel(\"$\\lambda$\")\n", - "plt.show()" - ] - }, - { - "cell_type": "code", - "execution_count": 20, - "metadata": { - "collapsed": false - }, - "outputs": [], - "source": [ - "# optional\n", - "# we can use log files to visualize our graph in Tensorboard\n", - "writer = tf.summary.FileWriter('logs/')\n", - "writer.add_graph(tf.get_default_graph())" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Using Keras\n", - "\n", - "Keras is a high level [neural network](https://en.wikipedia.org/wiki/Application_programming_interface)\n", - "that supports Tensorflow, CTNK and Theano as backends. \n", - "If you have Tensorflow installed Keras is available through the *tf.keras* module. \n", - "If you have Anaconda installed you may run the following command" - ] - }, - { - "cell_type": "code", - "execution_count": 21, - "metadata": { - "collapsed": false - }, - "outputs": [], - "source": [ - "conda install keras" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager:" - ] - }, - { - "cell_type": "code", - "execution_count": 22, - "metadata": { - "collapsed": false - }, - "outputs": [], - "source": [ - "pip install keras" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "or look up the [instructions here](https://keras.io/)." - ] - }, - { - "cell_type": "code", - "execution_count": 23, - "metadata": { - "collapsed": false - }, - "outputs": [], - "source": [ - "import tensorflow as tf\n", - "from tensorflow.keras.layers import Input\n", - "from tensorflow.keras.models import Sequential #This allows appending layers to existing models\n", - "from tensorflow.keras.layers import Dense #This allows defining the characteristics of a particular layer\n", - "from tensorflow.keras import optimizers #This allows using whichever optimiser we want (sgd,adam,RMSprop)\n", - "from tensorflow.keras import regularizers #This allows using whichever regularizer we want (l1,l2,l1_l2)\n", - "from tensorflow.keras.utils import to_categorical #This allows using categorical cross entropy as the cost function\n", "\n", "def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd):\n", " model = Sequential()\n", @@ -1849,7 +1611,7 @@ }, { "cell_type": "code", - "execution_count": 24, + "execution_count": 20, "metadata": { "collapsed": false }, @@ -1874,7 +1636,7 @@ }, { "cell_type": "code", - "execution_count": 25, + "execution_count": 21, "metadata": { "collapsed": false }, @@ -1922,7 +1684,7 @@ }, { "cell_type": "code", - "execution_count": 26, + "execution_count": 22, "metadata": { "collapsed": false }, @@ -2100,6 +1862,47 @@ "cell_type": "markdown", "metadata": {}, "source": [ + "## Fine-tuning neural network hyperparameters\n", + "\n", + "The flexibility of neural networks is also one of their main\n", + "drawbacks: there are many hyperparameters to tweak. Not only can you\n", + "use any imaginable network topology (how neurons/nodes are interconnected),\n", + "but even in a simple FFNN you can change the number of layers, the\n", + "number of neurons per layer, the type of activation function to use in\n", + "each layer, the weight initialization logic, the stochastic gradient optmized and much more. How do you\n", + "know what combination of hyperparameters is the best for your task?\n", + "\n", + "* You can use grid search with cross-validation to find the right hyperparameters.\n", + "\n", + "However,since there are many hyperparameters to tune, and since\n", + "training a neural network on a large dataset takes a lot of time, you\n", + "will only be able to explore a tiny part of the hyperparameter space.\n", + "\n", + "\n", + "* You can use randomized search.\n", + "\n", + "* Or use tools like [Oscar](http://oscar.calldesk.ai/), which implements more complex algorithms to help you find a good set of hyperparameters quickly. \n", + "\n", + "## Hidden layers\n", + "\n", + "\n", + "\n", + "For many problems you can start with just one or two hidden layers and it will work just fine.\n", + "For the MNIST data set you ca easily get a high accuracy using just one hidden layer with a\n", + "few hundred neurons.\n", + "You can reach for this data set above 98% accuracy using two hidden layers with the same total amount of\n", + "neurons, in roughly the same amount of training time. \n", + "\n", + "For more complex problems, you can gradually\n", + "ramp up the number of hidden layers, until you start overfitting the training set. Very complex tasks, such\n", + "as large image classification or speech recognition, typically require networks with dozens of layers\n", + "and they need a huge amount\n", + "of training data. However, you will rarely have to train such networks from scratch: it is much more\n", + "common to reuse parts of a pretrained state-of-the-art network that performs a similar task.\n", + "\n", + "\n", + "\n", + "\n", "\n", "## Which activation function should I use?\n", "\n", @@ -2125,6 +1928,19 @@ "neural networks suffer from unstable gradients, different layers may\n", "learn at widely different speeds\n", "\n", + "\n", + "## More on activation functions, output layers\n", + "\n", + "In most cases you can use the ReLU activation function in the hidden layers (or one of its variants).\n", + "\n", + "It is a bit faster to compute than other activation functions, and the gradient descent optimization does in general not get stuck.\n", + "\n", + "**For the output layer:**\n", + "\n", + "* For classification the softmax activation function is generally a good choice for classification tasks (when the classes are mutually exclusive).\n", + "\n", + "* For regression tasks, you can simply use no activation function at all.\n", + "\n", "\n", "## Is the Logistic activation function (Sigmoid) our choice?\n", "\n", @@ -2230,6 +2046,40 @@ "spare time and computing power, you can use cross-validation or\n", "bootstrap to evaluate other activation functions.\n", "\n", + "## Batch Normalization\n", + "\n", + "Batch Normalization\n", + "aims to address the vanishing/exploding gradients problems, and more generally the problem that the\n", + "distribution of each layer’s inputs changes during training, as the parameters of the previous layers change.\n", + "\n", + "The technique consists of adding an operation in the model just before the activation function of each\n", + "layer, simply zero-centering and normalizing the inputs, then scaling and shifting the result using two new\n", + "parameters per layer (one for scaling, the other for shifting). In other words, this operation lets the model\n", + "learn the optimal scale and mean of the inputs for each layer.\n", + "In order to zero-center and normalize the inputs, the algorithm needs to estimate the inputs’ mean and\n", + "standard deviation. It does so by evaluating the mean and standard deviation of the inputs over the current\n", + "mini-batch, from this the name batch normalization.\n", + "\n", + "## Dropout\n", + "\n", + "It is a fairly simple algorithm: at every training step, every neuron (including the input neurons but\n", + "excluding the output neurons) has a probability $p$ of being temporarily dropped out, meaning it will be\n", + "entirely ignored during this training step, but it may be active during the next step.\n", + "\n", + "The\n", + "hyperparameter $p$ is called the dropout rate, and it is typically set to 50%. After training, the neurons are not dropped anymore.\n", + " It is viewed as one of the most popular regularization techniques.\n", + "\n", + "## Gradient Clipping\n", + "\n", + "A popular technique to lessen the exploding gradients problem is to simply clip the gradients during\n", + "backpropagation so that they never exceed some threshold (this is mostly useful for recurrent neural\n", + "networks).\n", + "\n", + "This technique is called Gradient Clipping.\n", + "\n", + "In general however, Batch\n", + "Normalization is preferred.\n", "\n", "\n", "## A top-down perspective on Neural networks\n", @@ -2449,7 +2299,321 @@ "For more material on convolutional networks, we strongly recommend\n", "the course\n", "[IN5400 – Machine Learning for Image Analysis](https://www.uio.no/studier/emner/matnat/ifi/IN5400/index-eng.html)\n", - "and the slides of [CS231](http://cs231n.github.io/convolutional-networks/) which is taught at Stanford University (consistently ranked as one of the top computer science programs in the world). [Michael Nielsen's book is a must read, in particular chapter 6 which deals with CNNs](http://neuralnetworksanddeeplearning.com/chap6.html)." + "and the slides of [CS231](http://cs231n.github.io/convolutional-networks/) which is taught at Stanford University (consistently ranked as one of the top computer science programs in the world). [Michael Nielsen's book is a must read, in particular chapter 6 which deals with CNNs](http://neuralnetworksanddeeplearning.com/chap6.html).\n", + "\n", + "\n", + "\n", + "## CNNs in more detail, building convolutional neural networks in Tensorflow and Keras\n", + "\n", + "\n", + "As discussed above, CNNs are neural networks built from the assumption that the inputs\n", + "to the network are 2D images. This is important because the number of features or pixels in images\n", + "grows very fast with the image size, and an enormous number of weights and biases are needed in order to build an accurate network. \n", + "\n", + "As before, we still have our input, a hidden layer and an output. What's novel about convolutional networks\n", + "are the **convolutional** and **pooling** layers stacked in pairs between the input and the hidden layer.\n", + "In addition, the data is no longer represented as a 2D feature matrix, instead each input is a number of 2D\n", + "matrices, typically 1 for each color dimension (Red, Green, Blue). \n", + "\n", + "\n", + "## Setting it up\n", + "\n", + "It means that to represent the entire\n", + "dataset of images, we require a 4D matrix or **tensor**. This tensor has the dimensions:" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(n_{inputs},\\, n_{pixels, width},\\, n_{pixels, height},\\, depth) .\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## The MNIST dataset again\n", + "\n", + "The MNIST dataset consists of grayscale images with a pixel size of\n", + "$28\\times 28$, meaning we require $28 \\times 28 = 724$ weights to each\n", + "neuron in the first hidden layer.\n", + "\n", + "If we were to analyze images of size $128\\times 128$ we would require\n", + "$128 \\times 128 = 16384$ weights to each neuron. Even worse if we were\n", + "dealing with color images, as most images are, we have an image matrix\n", + "of size $128\\times 128$ for each color dimension (Red, Green, Blue),\n", + "meaning 3 times the number of weights $= 49152$ are required for every\n", + "single neuron in the first hidden layer.\n", + "\n", + "\n", + "## Strong correlations\n", + "\n", + "Images typically have strong local correlations, meaning that a small\n", + "part of the image varies little from its neighboring regions. If for\n", + "example we have an image of a blue car, we can roughly assume that a\n", + "small blue part of the image is surrounded by other blue regions.\n", + "\n", + "Therefore, instead of connecting every single pixel to a neuron in the\n", + "first hidden layer, as we have previously done with deep neural\n", + "networks, we can instead connect each neuron to a small part of the\n", + "image (in all 3 RGB depth dimensions). The size of each small area is\n", + "fixed, and known as a [receptive](https://en.wikipedia.org/wiki/Receptive_field).\n", + "\n", + "\n", + "\n", + "## Layers of a CNN\n", + "The layers of a convolutional neural network arrange neurons in 3D: width, height and depth. \n", + "The input image is typically a square matrix of depth 3. \n", + "\n", + "A **convolution** is performed on the image which outputs\n", + "a 3D volume of neurons. The weights to the input are arranged in a number of 2D matrices, known as **filters**.\n", + "\n", + "\n", + "Each filter slides along the input image, taking the dot product\n", + "between each small part of the image and the filter, in all depth\n", + "dimensions. This is then passed through a non-linear function,\n", + "typically the **Rectified Linear (ReLu)** function, which serves as the\n", + "activation of the neurons in the first convolutional layer. This is\n", + "further passed through a **pooling layer**, which reduces the size of the\n", + "convolutional layer, e.g. by taking the maximum or average across some\n", + "small regions, and this serves as input to the next convolutional\n", + "layer.\n", + "\n", + "\n", + "## Systematic reduction\n", + "\n", + "By systematically reducing the size of the input volume, through\n", + "convolution and pooling, the network should create representations of\n", + "small parts of the input, and then from them assemble representations\n", + "of larger areas. The final pooling layer is flattened to serve as\n", + "input to a hidden layer, such that each neuron in the final pooling\n", + "layer is connected to every single neuron in the hidden layer. This\n", + "then serves as input to the output layer, e.g. a softmax output for\n", + "classification.\n", + "\n", + "\n", + "## Prerequisites: Collect and pre-process data" + ] + }, + { + "cell_type": "code", + "execution_count": 23, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "# import necessary packages\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn import datasets\n", + "\n", + "\n", + "# ensure the same random numbers appear every time\n", + "np.random.seed(0)\n", + "\n", + "# display images in notebook\n", + "%matplotlib inline\n", + "plt.rcParams['figure.figsize'] = (12,12)\n", + "\n", + "\n", + "# download MNIST dataset\n", + "digits = datasets.load_digits()\n", + "\n", + "# define inputs and labels\n", + "inputs = digits.images\n", + "labels = digits.target\n", + "\n", + "# RGB images have a depth of 3\n", + "# our images are grayscale so they should have a depth of 1\n", + "inputs = inputs[:,:,:,np.newaxis]\n", + "\n", + "print(\"inputs = (n_inputs, pixel_width, pixel_height, depth) = \" + str(inputs.shape))\n", + "print(\"labels = (n_inputs) = \" + str(labels.shape))\n", + "\n", + "\n", + "# choose some random images to display\n", + "n_inputs = len(inputs)\n", + "indices = np.arange(n_inputs)\n", + "random_indices = np.random.choice(indices, size=5)\n", + "\n", + "for i, image in enumerate(digits.images[random_indices]):\n", + " plt.subplot(1, 5, i+1)\n", + " plt.axis('off')\n", + " plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest')\n", + " plt.title(\"Label: %d\" % digits.target[random_indices[i]])\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Importing Keras and Tensorflow" + ] + }, + { + "cell_type": "code", + "execution_count": 24, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "from keras.utils import to_categorical\n", + "from sklearn.model_selection import train_test_split\n", + "\n", + "# representation of labels\n", + "labels = to_categorical(labels)\n", + "\n", + "# split into train and test data\n", + "# one-liner from scikit-learn library\n", + "train_size = 0.8\n", + "test_size = 1 - train_size\n", + "X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size,\n", + " test_size=test_size)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "## Running with Keras" + ] + }, + { + "cell_type": "code", + "execution_count": 25, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "from keras.models import Sequential\n", + "from keras.layers.convolutional import Conv2D\n", + "from keras.layers.convolutional import MaxPooling2D\n", + "from keras.layers import Flatten\n", + "from keras.layers import Dense\n", + "from keras.regularizers import l2\n", + "from keras.optimizers import SGD\n", + "\n", + "def create_convolutional_neural_network_keras(input_shape, receptive_field,\n", + " n_filters, n_neurons_connected, n_categories,\n", + " eta, lmbd):\n", + " model = Sequential()\n", + " model.add(Conv2D(n_filters, (receptive_field, receptive_field), input_shape=input_shape, padding='same',\n", + " activation='relu', kernel_regularizer=l2(lmbd)))\n", + " model.add(MaxPooling2D(pool_size=(2, 2)))\n", + " model.add(Flatten())\n", + " model.add(Dense(n_neurons_connected, activation='relu', kernel_regularizer=l2(lmbd)))\n", + " model.add(Dense(n_categories, activation='softmax', kernel_regularizer=l2(lmbd)))\n", + " \n", + " sgd = SGD(lr=eta)\n", + " model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy'])\n", + " \n", + " return model\n", + "\n", + "epochs = 100\n", + "batch_size = 100\n", + "input_shape = X_train.shape[1:4]\n", + "receptive_field = 3\n", + "n_filters = 10\n", + "n_neurons_connected = 50\n", + "n_categories = 10\n", + "\n", + "eta_vals = np.logspace(-5, 1, 7)\n", + "lmbd_vals = np.logspace(-5, 1, 7)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Final part" + ] + }, + { + "cell_type": "code", + "execution_count": 26, + "metadata": { + "collapsed": false + }, + "outputs": [], + "source": [ + "CNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)\n", + " \n", + "for i, eta in enumerate(eta_vals):\n", + " for j, lmbd in enumerate(lmbd_vals):\n", + " CNN = create_convolutional_neural_network_keras(input_shape, receptive_field,\n", + " n_filters, n_neurons_connected, n_categories,\n", + " eta, lmbd)\n", + " CNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0)\n", + " scores = CNN.evaluate(X_test, Y_test)\n", + " \n", + " CNN_keras[i][j] = CNN\n", + " \n", + " print(\"Learning rate = \", eta)\n", + " print(\"Lambda = \", lmbd)\n", + " print(\"Test accuracy: %.3f\" % scores[1])\n", + " print()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Final visualization" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + " # visual representation of grid search\n", + " # uses seaborn heatmap, could probably do this in matplotlib\n", + " import seaborn as sns\n", + " \n", + " sns.set()\n", + " \n", + " train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))\n", + " test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))\n", + " \n", + " for i in range(len(eta_vals)):\n", + " for j in range(len(lmbd_vals)):\n", + " CNN = CNN_keras[i][j]\n", + " \n", + " train_accuracy[i][j] = CNN.evaluate(X_train, Y_train)[1]\n", + " test_accuracy[i][j] = CNN.evaluate(X_test, Y_test)[1]\n", + " \n", + " \n", + " fig, ax = plt.subplots(figsize = (10, 10))\n", + " sns.heatmap(train_accuracy, annot=True, ax=ax, cmap=\"viridis\")\n", + " ax.set_title(\"Training Accuracy\")\n", + " ax.set_ylabel(\"$\\eta$\")\n", + " ax.set_xlabel(\"$\\lambda$\")\n", + " plt.show()\n", + " \n", + " fig, ax = plt.subplots(figsize = (10, 10))\n", + " sns.heatmap(test_accuracy, annot=True, ax=ax, cmap=\"viridis\")\n", + " ax.set_title(\"Test Accuracy\")\n", + " ax.set_ylabel(\"$\\eta$\")\n", + " ax.set_xlabel(\"$\\lambda$\")\n", + " plt.show()\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Fun links\n", + "\n", + "1. [Self-Driving cars using a convolutional neural network](https://arxiv.org/abs/1604.07316)\n", + "\n", + "2. [Abstract art using convolutional neural networks](https://deepdreamgenerator.com/)" ] } ], diff --git a/doc/src/week41/week41.do.txt b/doc/src/week41/week41.do.txt index 3bc9371bc..026c7ba06 100644 --- a/doc/src/week41/week41.do.txt +++ b/doc/src/week41/week41.do.txt @@ -1100,9 +1100,35 @@ To install tensorflow on Unix/Linux systems, use pip as pip3 install tensorflow !ec and/or if you use _anaconda_, just write (or install from the graphical user interface) -!bc pycod -conda install tensorflow +(current release of CPU-only TensorFlow) +!bc pycod +conda create -n tf tensorflow +conda activate tf !ec +To install the current release of GPU TensorFlow +!bc pycod +conda create -n tf-gpu tensorflow-gpu +conda activate tf-gpu +!ec + +!split +===== Using Keras ===== + +Keras is a high level "neural network":"https://en.wikipedia.org/wiki/Application_programming_interface" +that supports Tensorflow, CTNK and Theano as backends. +If you have Tensorflow installed Keras is available through the *tf.keras* module. +If you have Anaconda installed you may run the following command +!bc pycod +conda install keras +!ec + +Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager: + +!bc pycod +pip install keras +!ec +or look up the "instructions here":"https://keras.io/". + !split ===== Collect and pre-process data ===== @@ -1111,6 +1137,7 @@ conda install tensorflow # import necessary packages import numpy as np import matplotlib.pyplot as plt +import tensorflow as tf from sklearn import datasets @@ -1153,7 +1180,13 @@ plt.show() !ec !bc pycod -from keras.utils import to_categorical +from tensorflow.keras.layers import Input +from tensorflow.keras.models import Sequential #This allows appending layers to existing models +from tensorflow.keras.layers import Dense #This allows defining the characteristics of a particular layer +from tensorflow.keras import optimizers #This allows using whichever optimiser we want (sgd,adam,RMSprop) +from tensorflow.keras import regularizers #This allows using whichever regularizer we want (l1,l2,l1_l2) +from tensorflow.keras.utils import to_categorical #This allows using categorical cross entropy as the cost function + from sklearn.model_selection import train_test_split # one-hot representation of labels @@ -1166,245 +1199,9 @@ X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=t test_size=test_size) !ec -!split -===== Using TensorFlow backend ===== - -o Define model and architecture -o Choose cost function and optimizer - -!bc pycod -import tensorflow as tf - -class NeuralNetworkTensorflow: - def __init__( - self, - X_train, - Y_train, - X_test, - Y_test, - n_neurons_layer1=100, - n_neurons_layer2=50, - n_categories=2, - epochs=10, - batch_size=100, - eta=0.1, - lmbd=0.0): - - # keep track of number of steps - self.global_step = tf.Variable(0, dtype=tf.int32, trainable=False, name='global_step') - - self.X_train = X_train - self.Y_train = Y_train - self.X_test = X_test - self.Y_test = Y_test - - self.n_inputs = X_train.shape[0] - self.n_features = X_train.shape[1] - self.n_neurons_layer1 = n_neurons_layer1 - self.n_neurons_layer2 = n_neurons_layer2 - self.n_categories = n_categories - - self.epochs = epochs - self.batch_size = batch_size - self.iterations = self.n_inputs // self.batch_size - self.eta = eta - self.lmbd = lmbd - - # build network piece by piece - # name scopes (with) are used to enforce creation of new variables - # https://www.tensorflow.org/guide/variables - self.create_placeholders() - self.create_DNN() - self.create_loss() - self.create_optimiser() - self.create_accuracy() - - def create_placeholders(self): - # placeholders are fine here, but "Datasets" are the preferred method - # of streaming data into a model - with tf.name_scope('data'): - self.X = tf.placeholder(tf.float32, shape=(None, self.n_features), name='X_data') - self.Y = tf.placeholder(tf.float32, shape=(None, self.n_categories), name='Y_data') - - def create_DNN(self): - with tf.name_scope('DNN'): - # the weights are stored to calculate regularization loss later - - # Fully connected layer 1 - self.W_fc1 = self.weight_variable([self.n_features, self.n_neurons_layer1], name='fc1', dtype=tf.float32) - b_fc1 = self.bias_variable([self.n_neurons_layer1], name='fc1', dtype=tf.float32) - a_fc1 = tf.nn.sigmoid(tf.matmul(self.X, self.W_fc1) + b_fc1) - - # Fully connected layer 2 - self.W_fc2 = self.weight_variable([self.n_neurons_layer1, self.n_neurons_layer2], name='fc2', dtype=tf.float32) - b_fc2 = self.bias_variable([self.n_neurons_layer2], name='fc2', dtype=tf.float32) - a_fc2 = tf.nn.sigmoid(tf.matmul(a_fc1, self.W_fc2) + b_fc2) - - # Output layer - self.W_out = self.weight_variable([self.n_neurons_layer2, self.n_categories], name='out', dtype=tf.float32) - b_out = self.bias_variable([self.n_categories], name='out', dtype=tf.float32) - self.z_out = tf.matmul(a_fc2, self.W_out) + b_out - - def create_loss(self): - with tf.name_scope('loss'): - softmax_loss = tf.reduce_mean(tf.nn.softmax_cross_entropy_with_logits_v2(labels=self.Y, logits=self.z_out)) - - regularizer_loss_fc1 = tf.nn.l2_loss(self.W_fc1) - regularizer_loss_fc2 = tf.nn.l2_loss(self.W_fc2) - regularizer_loss_out = tf.nn.l2_loss(self.W_out) - regularizer_loss = self.lmbd*(regularizer_loss_fc1 + regularizer_loss_fc2 + regularizer_loss_out) - - self.loss = softmax_loss + regularizer_loss - - def create_accuracy(self): - with tf.name_scope('accuracy'): - probabilities = tf.nn.softmax(self.z_out) - predictions = tf.argmax(probabilities, axis=1) - labels = tf.argmax(self.Y, axis=1) - - correct_predictions = tf.equal(predictions, labels) - correct_predictions = tf.cast(correct_predictions, tf.float32) - self.accuracy = tf.reduce_mean(correct_predictions) - - def create_optimiser(self): - with tf.name_scope('optimizer'): - self.optimizer = tf.train.GradientDescentOptimizer(learning_rate=self.eta).minimize(self.loss, global_step=self.global_step) - - def weight_variable(self, shape, name='', dtype=tf.float32): - initial = tf.truncated_normal(shape, stddev=0.1) - return tf.Variable(initial, name=name, dtype=dtype) - - def bias_variable(self, shape, name='', dtype=tf.float32): - initial = tf.constant(0.1, shape=shape) - return tf.Variable(initial, name=name, dtype=dtype) - - def fit(self): - data_indices = np.arange(self.n_inputs) - - with tf.Session() as sess: - sess.run(tf.global_variables_initializer()) - for i in range(self.epochs): - for j in range(self.iterations): - chosen_datapoints = np.random.choice(data_indices, size=self.batch_size, replace=False) - batch_X, batch_Y = self.X_train[chosen_datapoints], self.Y_train[chosen_datapoints] - - sess.run([DNN.loss, DNN.optimizer], - feed_dict={DNN.X: batch_X, - DNN.Y: batch_Y}) - accuracy = sess.run(DNN.accuracy, - feed_dict={DNN.X: batch_X, - DNN.Y: batch_Y}) - step = sess.run(DNN.global_step) - - self.train_loss, self.train_accuracy = sess.run([DNN.loss, DNN.accuracy], - feed_dict={DNN.X: self.X_train, - DNN.Y: self.Y_train}) - - self.test_loss, self.test_accuracy = sess.run([DNN.loss, DNN.accuracy], - feed_dict={DNN.X: self.X_test, - DNN.Y: self.Y_test}) -!ec - - -!split -===== Optimizing and using gradient descent ===== - -!bc pycod -epochs = 100 -batch_size = 100 -n_neurons_layer1 = 100 -n_neurons_layer2 = 50 -n_categories = 10 -eta_vals = np.logspace(-5, 1, 7) -lmbd_vals = np.logspace(-5, 1, 7) -!ec !bc pycod -DNN_tf = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object) - -for i, eta in enumerate(eta_vals): - for j, lmbd in enumerate(lmbd_vals): - DNN = NeuralNetworkTensorflow(X_train, Y_train, X_test, Y_test, - n_neurons_layer1, n_neurons_layer2, n_categories, - epochs=epochs, batch_size=batch_size, eta=eta, lmbd=lmbd) - DNN.fit() - - DNN_tf[i][j] = DNN - - print("Learning rate = ", eta) - print("Lambda = ", lmbd) - print("Test accuracy: %.3f" % DNN.test_accuracy) - print() -!ec - -!bc pycod -# optional -# visual representation of grid search -# uses seaborn heatmap, could probably do this in matplotlib -import seaborn as sns - -sns.set() - -train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals))) -test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals))) - -for i in range(len(eta_vals)): - for j in range(len(lmbd_vals)): - DNN = DNN_tf[i][j] - - train_accuracy[i][j] = DNN.train_accuracy - test_accuracy[i][j] = DNN.test_accuracy - - -fig, ax = plt.subplots(figsize = (10, 10)) -sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis") -ax.set_title("Training Accuracy") -ax.set_ylabel("$\eta$") -ax.set_xlabel("$\lambda$") -plt.show() - -fig, ax = plt.subplots(figsize = (10, 10)) -sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis") -ax.set_title("Test Accuracy") -ax.set_ylabel("$\eta$") -ax.set_xlabel("$\lambda$") -plt.show() -!ec - -!bc pycod -# optional -# we can use log files to visualize our graph in Tensorboard -writer = tf.summary.FileWriter('logs/') -writer.add_graph(tf.get_default_graph()) -!ec - -!split -===== Using Keras ===== - -Keras is a high level "neural network":"https://en.wikipedia.org/wiki/Application_programming_interface" -that supports Tensorflow, CTNK and Theano as backends. -If you have Tensorflow installed Keras is available through the *tf.keras* module. -If you have Anaconda installed you may run the following command -!bc pycod -conda install keras -!ec - -Alternatively, if you have Tensorflow or one of the other supported backends install you may use the pip package manager: - -!bc pycod -pip install keras -!ec -or look up the "instructions here":"https://keras.io/". - -!bc pycod -import tensorflow as tf -from tensorflow.keras.layers import Input -from tensorflow.keras.models import Sequential #This allows appending layers to existing models -from tensorflow.keras.layers import Dense #This allows defining the characteristics of a particular layer -from tensorflow.keras import optimizers #This allows using whichever optimiser we want (sgd,adam,RMSprop) -from tensorflow.keras import regularizers #This allows using whichever regularizer we want (l1,l2,l1_l2) -from tensorflow.keras.utils import to_categorical #This allows using categorical cross entropy as the cost function def create_neural_network_keras(n_neurons_layer1, n_neurons_layer2, n_categories, eta, lmbd): model = Sequential() @@ -1648,6 +1445,47 @@ plot_data(eta,n_neuron,Test_accuracy, 'testing') !ec +!split +===== Fine-tuning neural network hyperparameters ===== + +The flexibility of neural networks is also one of their main +drawbacks: there are many hyperparameters to tweak. Not only can you +use any imaginable network topology (how neurons/nodes are interconnected), +but even in a simple FFNN you can change the number of layers, the +number of neurons per layer, the type of activation function to use in +each layer, the weight initialization logic, the stochastic gradient optmized and much more. How do you +know what combination of hyperparameters is the best for your task? + +* You can use grid search with cross-validation to find the right hyperparameters. + +However,since there are many hyperparameters to tune, and since +training a neural network on a large dataset takes a lot of time, you +will only be able to explore a tiny part of the hyperparameter space. + + +* You can use randomized search. +* Or use tools like "Oscar":"http://oscar.calldesk.ai/", which implements more complex algorithms to help you find a good set of hyperparameters quickly. + +!split +===== Hidden layers ===== + + + +For many problems you can start with just one or two hidden layers and it will work just fine. +For the MNIST data set you ca easily get a high accuracy using just one hidden layer with a +few hundred neurons. +You can reach for this data set above 98% accuracy using two hidden layers with the same total amount of +neurons, in roughly the same amount of training time. + +For more complex problems, you can gradually +ramp up the number of hidden layers, until you start overfitting the training set. Very complex tasks, such +as large image classification or speech recognition, typically require networks with dozens of layers +and they need a huge amount +of training data. However, you will rarely have to train such networks from scratch: it is much more +common to reuse parts of a pretrained state-of-the-art network that performs a similar task. + + + !split ===== Which activation function should I use? ===== @@ -1674,6 +1512,19 @@ mostly encountered in recurrent neural networks. More generally, deep neural networks suffer from unstable gradients, different layers may learn at widely different speeds + +!split +===== More on activation functions, output layers ===== + +In most cases you can use the ReLU activation function in the hidden layers (or one of its variants). + +It is a bit faster to compute than other activation functions, and the gradient descent optimization does in general not get stuck. + +_For the output layer:_ + +* For classification the softmax activation function is generally a good choice for classification tasks (when the classes are mutually exclusive). +* For regression tasks, you can simply use no activation function at all. + !split ===== Is the Logistic activation function (Sigmoid) our choice? ===== @@ -1775,6 +1626,43 @@ $\alpha$ of $0.01$ for the leaky ReLU, and $1$ for ELU. If you have spare time and computing power, you can use cross-validation or bootstrap to evaluate other activation functions. +!split +===== Batch Normalization ===== + +Batch Normalization +aims to address the vanishing/exploding gradients problems, and more generally the problem that the +distribution of each layer’s inputs changes during training, as the parameters of the previous layers change. + +The technique consists of adding an operation in the model just before the activation function of each +layer, simply zero-centering and normalizing the inputs, then scaling and shifting the result using two new +parameters per layer (one for scaling, the other for shifting). In other words, this operation lets the model +learn the optimal scale and mean of the inputs for each layer. +In order to zero-center and normalize the inputs, the algorithm needs to estimate the inputs’ mean and +standard deviation. It does so by evaluating the mean and standard deviation of the inputs over the current +mini-batch, from this the name batch normalization. + +!split +===== Dropout ===== + +It is a fairly simple algorithm: at every training step, every neuron (including the input neurons but +excluding the output neurons) has a probability $p$ of being temporarily dropped out, meaning it will be +entirely ignored during this training step, but it may be active during the next step. + +The +hyperparameter $p$ is called the dropout rate, and it is typically set to 50%. After training, the neurons are not dropped anymore. + It is viewed as one of the most popular regularization techniques. + +!split +===== Gradient Clipping ===== + +A popular technique to lessen the exploding gradients problem is to simply clip the gradients during +backpropagation so that they never exceed some threshold (this is mostly useful for recurrent neural +networks). + +This technique is called Gradient Clipping. + +In general however, Batch +Normalization is preferred. !split ===== A top-down perspective on Neural networks ===== @@ -1979,3 +1867,261 @@ the course and the slides of "CS231":"http://cs231n.github.io/convolutional-networks/" which is taught at Stanford University (consistently ranked as one of the top computer science programs in the world). "Michael Nielsen's book is a must read, in particular chapter 6 which deals with CNNs":"http://neuralnetworksanddeeplearning.com/chap6.html". + +!split +===== CNNs in more detail, building convolutional neural networks in Tensorflow and Keras ===== + + +As discussed above, CNNs are neural networks built from the assumption that the inputs +to the network are 2D images. This is important because the number of features or pixels in images +grows very fast with the image size, and an enormous number of weights and biases are needed in order to build an accurate network. + +As before, we still have our input, a hidden layer and an output. What's novel about convolutional networks +are the _convolutional_ and _pooling_ layers stacked in pairs between the input and the hidden layer. +In addition, the data is no longer represented as a 2D feature matrix, instead each input is a number of 2D +matrices, typically 1 for each color dimension (Red, Green, Blue). + + +!split +===== Setting it up ===== + +It means that to represent the entire +dataset of images, we require a 4D matrix or _tensor_. This tensor has the dimensions: +!bt +\[ +(n_{inputs},\, n_{pixels, width},\, n_{pixels, height},\, depth) . +\] +!et + +!split +===== The MNIST dataset again ===== + +The MNIST dataset consists of grayscale images with a pixel size of +$28\times 28$, meaning we require $28 \times 28 = 724$ weights to each +neuron in the first hidden layer. + +If we were to analyze images of size $128\times 128$ we would require +$128 \times 128 = 16384$ weights to each neuron. Even worse if we were +dealing with color images, as most images are, we have an image matrix +of size $128\times 128$ for each color dimension (Red, Green, Blue), +meaning 3 times the number of weights $= 49152$ are required for every +single neuron in the first hidden layer. + + +!split +===== Strong correlations ===== + +Images typically have strong local correlations, meaning that a small +part of the image varies little from its neighboring regions. If for +example we have an image of a blue car, we can roughly assume that a +small blue part of the image is surrounded by other blue regions. + +Therefore, instead of connecting every single pixel to a neuron in the +first hidden layer, as we have previously done with deep neural +networks, we can instead connect each neuron to a small part of the +image (in all 3 RGB depth dimensions). The size of each small area is +fixed, and known as a "receptive":"https://en.wikipedia.org/wiki/Receptive_field". + + +!split +===== Layers of a CNN ===== +The layers of a convolutional neural network arrange neurons in 3D: width, height and depth. +The input image is typically a square matrix of depth 3. + +A _convolution_ is performed on the image which outputs +a 3D volume of neurons. The weights to the input are arranged in a number of 2D matrices, known as _filters_. + + +Each filter slides along the input image, taking the dot product +between each small part of the image and the filter, in all depth +dimensions. This is then passed through a non-linear function, +typically the _Rectified Linear (ReLu)_ function, which serves as the +activation of the neurons in the first convolutional layer. This is +further passed through a _pooling layer_, which reduces the size of the +convolutional layer, e.g. by taking the maximum or average across some +small regions, and this serves as input to the next convolutional +layer. + + +!split +===== Systematic reduction ===== + +By systematically reducing the size of the input volume, through +convolution and pooling, the network should create representations of +small parts of the input, and then from them assemble representations +of larger areas. The final pooling layer is flattened to serve as +input to a hidden layer, such that each neuron in the final pooling +layer is connected to every single neuron in the hidden layer. This +then serves as input to the output layer, e.g. a softmax output for +classification. + + +!split +===== Prerequisites: Collect and pre-process data ===== +!bc pycod +# import necessary packages +import numpy as np +import matplotlib.pyplot as plt +from sklearn import datasets + + +# ensure the same random numbers appear every time +np.random.seed(0) + +# display images in notebook +%matplotlib inline +plt.rcParams['figure.figsize'] = (12,12) + + +# download MNIST dataset +digits = datasets.load_digits() + +# define inputs and labels +inputs = digits.images +labels = digits.target + +# RGB images have a depth of 3 +# our images are grayscale so they should have a depth of 1 +inputs = inputs[:,:,:,np.newaxis] + +print("inputs = (n_inputs, pixel_width, pixel_height, depth) = " + str(inputs.shape)) +print("labels = (n_inputs) = " + str(labels.shape)) + + +# choose some random images to display +n_inputs = len(inputs) +indices = np.arange(n_inputs) +random_indices = np.random.choice(indices, size=5) + +for i, image in enumerate(digits.images[random_indices]): + plt.subplot(1, 5, i+1) + plt.axis('off') + plt.imshow(image, cmap=plt.cm.gray_r, interpolation='nearest') + plt.title("Label: %d" % digits.target[random_indices[i]]) +plt.show() +!ec + + +!split +===== Importing Keras and Tensorflow ===== +!bc pycod +from keras.utils import to_categorical +from sklearn.model_selection import train_test_split + +# representation of labels +labels = to_categorical(labels) + +# split into train and test data +# one-liner from scikit-learn library +train_size = 0.8 +test_size = 1 - train_size +X_train, X_test, Y_train, Y_test = train_test_split(inputs, labels, train_size=train_size, + test_size=test_size) +!ec + +!split +===== Running with Keras ===== + +!bc pycod +from keras.models import Sequential +from keras.layers.convolutional import Conv2D +from keras.layers.convolutional import MaxPooling2D +from keras.layers import Flatten +from keras.layers import Dense +from keras.regularizers import l2 +from keras.optimizers import SGD + +def create_convolutional_neural_network_keras(input_shape, receptive_field, + n_filters, n_neurons_connected, n_categories, + eta, lmbd): + model = Sequential() + model.add(Conv2D(n_filters, (receptive_field, receptive_field), input_shape=input_shape, padding='same', + activation='relu', kernel_regularizer=l2(lmbd))) + model.add(MaxPooling2D(pool_size=(2, 2))) + model.add(Flatten()) + model.add(Dense(n_neurons_connected, activation='relu', kernel_regularizer=l2(lmbd))) + model.add(Dense(n_categories, activation='softmax', kernel_regularizer=l2(lmbd))) + + sgd = SGD(lr=eta) + model.compile(loss='categorical_crossentropy', optimizer=sgd, metrics=['accuracy']) + + return model + +epochs = 100 +batch_size = 100 +input_shape = X_train.shape[1:4] +receptive_field = 3 +n_filters = 10 +n_neurons_connected = 50 +n_categories = 10 + +eta_vals = np.logspace(-5, 1, 7) +lmbd_vals = np.logspace(-5, 1, 7) +!ec + +!split +===== Final part ===== + +!bc pycod +CNN_keras = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object) + +for i, eta in enumerate(eta_vals): + for j, lmbd in enumerate(lmbd_vals): + CNN = create_convolutional_neural_network_keras(input_shape, receptive_field, + n_filters, n_neurons_connected, n_categories, + eta, lmbd) + CNN.fit(X_train, Y_train, epochs=epochs, batch_size=batch_size, verbose=0) + scores = CNN.evaluate(X_test, Y_test) + + CNN_keras[i][j] = CNN + + print("Learning rate = ", eta) + print("Lambda = ", lmbd) + print("Test accuracy: %.3f" % scores[1]) + print() +!ec + +!split +===== Final visualization ===== + +!bc +# visual representation of grid search +# uses seaborn heatmap, could probably do this in matplotlib +import seaborn as sns + +sns.set() + +train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals))) +test_accuracy = np.zeros((len(eta_vals), len(lmbd_vals))) + +for i in range(len(eta_vals)): + for j in range(len(lmbd_vals)): + CNN = CNN_keras[i][j] + + train_accuracy[i][j] = CNN.evaluate(X_train, Y_train)[1] + test_accuracy[i][j] = CNN.evaluate(X_test, Y_test)[1] + + +fig, ax = plt.subplots(figsize = (10, 10)) +sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis") +ax.set_title("Training Accuracy") +ax.set_ylabel("$\eta$") +ax.set_xlabel("$\lambda$") +plt.show() + +fig, ax = plt.subplots(figsize = (10, 10)) +sns.heatmap(test_accuracy, annot=True, ax=ax, cmap="viridis") +ax.set_title("Test Accuracy") +ax.set_ylabel("$\eta$") +ax.set_xlabel("$\lambda$") +plt.show() +!ec + +!split +===== Fun links ===== + +o "Self-Driving cars using a convolutional neural network":"https://arxiv.org/abs/1604.07316" +o "Abstract art using convolutional neural networks":"https://deepdreamgenerator.com/" + + +