421 lines
26 KiB
HTML
421 lines
26 KiB
HTML
<!--
|
||
Automatically generated HTML file from DocOnce source
|
||
(https://github.com/hplgit/doconce/)
|
||
-->
|
||
<html>
|
||
<head>
|
||
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
||
<meta name="generator" content="DocOnce: https://github.com/hplgit/doconce/" />
|
||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||
<meta name="description" content="Week 41 Tensor flow and Deep Learning, Convolutional Neural Networks">
|
||
|
||
<title>Week 41 Tensor flow and Deep Learning, Convolutional Neural Networks</title>
|
||
|
||
<!-- Bootstrap style: bootstrap -->
|
||
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
|
||
<!-- not necessary
|
||
<link href="https://netdna.bootstrapcdn.com/font-awesome/4.0.3/css/font-awesome.css" rel="stylesheet">
|
||
-->
|
||
|
||
<style type="text/css">
|
||
|
||
/* Add scrollbar to dropdown menus in bootstrap navigation bar */
|
||
.dropdown-menu {
|
||
height: auto;
|
||
max-height: 400px;
|
||
overflow-x: hidden;
|
||
}
|
||
|
||
/* Adds an invisible element before each target to offset for the navigation
|
||
bar */
|
||
.anchor::before {
|
||
content:"";
|
||
display:block;
|
||
height:50px; /* fixed header height for style bootstrap */
|
||
margin:-50px 0 0; /* negative fixed header height */
|
||
}
|
||
</style>
|
||
|
||
|
||
</head>
|
||
|
||
<!-- tocinfo
|
||
{'highest level': 2,
|
||
'sections': [('Plan for week 41', 2, None, '___sec0'),
|
||
('Setting up the Back propagation algorithm', 2, None, '___sec1'),
|
||
('Setting up a Multi-layer perceptron model for classification',
|
||
2,
|
||
None,
|
||
'___sec2'),
|
||
('Defining the cost function', 2, None, '___sec3'),
|
||
('Example: binary classification problem', 2, None, '___sec4'),
|
||
('The Softmax function', 2, None, '___sec5'),
|
||
('Developing a code for doing neural networks with back '
|
||
'propagation',
|
||
2,
|
||
None,
|
||
'___sec6'),
|
||
('Collect and pre-process data', 2, None, '___sec7'),
|
||
('Train and test datasets', 2, None, '___sec8'),
|
||
('Define model and architecture', 2, None, '___sec9'),
|
||
('Layers', 2, None, '___sec10'),
|
||
('Weights and biases', 2, None, '___sec11'),
|
||
('Feed-forward pass', 2, None, '___sec12'),
|
||
('Matrix multiplications', 2, None, '___sec13'),
|
||
('Choose cost function and optimizer', 2, None, '___sec14'),
|
||
('Optimizing the cost function', 2, None, '___sec15'),
|
||
('Regularization', 2, None, '___sec16'),
|
||
('Matrix multiplication', 2, None, '___sec17'),
|
||
('Improving performance', 2, None, '___sec18'),
|
||
('Full object-oriented implementation', 2, None, '___sec19'),
|
||
('Evaluate model performance on test data', 2, None, '___sec20'),
|
||
('Adjust hyperparameters', 2, None, '___sec21'),
|
||
('Visualization', 2, None, '___sec22'),
|
||
('scikit-learn implementation', 2, None, '___sec23'),
|
||
('Visualization', 2, None, '___sec24'),
|
||
('Building neural networks in Tensorflow and Keras',
|
||
2,
|
||
None,
|
||
'___sec25'),
|
||
('Tensorflow', 2, None, '___sec26'),
|
||
('Using Keras', 2, None, '___sec27'),
|
||
('Collect and pre-process data', 2, None, '___sec28'),
|
||
('The Breast Cancer Data, now with Keras', 2, None, '___sec29'),
|
||
('Fine-tuning neural network hyperparameters',
|
||
2,
|
||
None,
|
||
'___sec30'),
|
||
('Hidden layers', 2, None, '___sec31'),
|
||
('Which activation function should I use?', 2, None, '___sec32'),
|
||
('Is the Logistic activation function (Sigmoid) our choice?',
|
||
2,
|
||
None,
|
||
'___sec33'),
|
||
('The derivative of the Logistic funtion', 2, None, '___sec34'),
|
||
('The RELU function family', 2, None, '___sec35'),
|
||
('Which activation function should we use?', 2, None, '___sec36'),
|
||
('More on activation functions, output layers',
|
||
2,
|
||
None,
|
||
'___sec37'),
|
||
('Batch Normalization', 2, None, '___sec38'),
|
||
('Dropout', 2, None, '___sec39'),
|
||
('Gradient Clipping', 2, None, '___sec40'),
|
||
('A very nice website on Neural Networks', 2, None, '___sec41'),
|
||
('A top-down perspective on Neural networks',
|
||
2,
|
||
None,
|
||
'___sec42'),
|
||
('Limitations of supervised learning with deep networks',
|
||
2,
|
||
None,
|
||
'___sec43'),
|
||
('Convolutional Neural Networks (recognizing images)',
|
||
2,
|
||
None,
|
||
'___sec44'),
|
||
('Regular NNs don’t scale well to full images',
|
||
2,
|
||
None,
|
||
'___sec45'),
|
||
('3D volumes of neurons', 2, None, '___sec46'),
|
||
('Layers used to build CNNs', 2, None, '___sec47'),
|
||
('Transforming images', 2, None, '___sec48'),
|
||
('CNNs in brief', 2, None, '___sec49'),
|
||
('CNNs in more detail, building convolutional neural networks in '
|
||
'Tensorflow and Keras',
|
||
2,
|
||
None,
|
||
'___sec50'),
|
||
('Setting it up', 2, None, '___sec51'),
|
||
('The MNIST dataset again', 2, None, '___sec52'),
|
||
('Strong correlations', 2, None, '___sec53'),
|
||
('Layers of a CNN', 2, None, '___sec54'),
|
||
('Systematic reduction', 2, None, '___sec55'),
|
||
('Prerequisites: Collect and pre-process data',
|
||
2,
|
||
None,
|
||
'___sec56'),
|
||
('Importing Keras and Tensorflow', 2, None, '___sec57'),
|
||
('Running with Keras', 2, None, '___sec58'),
|
||
('Final part', 2, None, '___sec59'),
|
||
('Final visualization', 2, None, '___sec60'),
|
||
('Fun links', 2, None, '___sec61')]}
|
||
end of tocinfo -->
|
||
|
||
<body>
|
||
|
||
|
||
|
||
<script type="text/x-mathjax-config">
|
||
MathJax.Hub.Config({
|
||
TeX: {
|
||
equationNumbers: { autoNumber: "none" },
|
||
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
|
||
}
|
||
});
|
||
</script>
|
||
<script type="text/javascript" async
|
||
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
|
||
</script>
|
||
|
||
|
||
|
||
|
||
<!-- Bootstrap navigation bar -->
|
||
<div class="navbar navbar-default navbar-fixed-top">
|
||
<div class="navbar-header">
|
||
<button type="button" class="navbar-toggle" data-toggle="collapse" data-target=".navbar-responsive-collapse">
|
||
<span class="icon-bar"></span>
|
||
<span class="icon-bar"></span>
|
||
<span class="icon-bar"></span>
|
||
</button>
|
||
<a class="navbar-brand" href="week41-bs.html">Week 41 Tensor flow and Deep Learning, Convolutional Neural Networks</a>
|
||
</div>
|
||
|
||
<div class="navbar-collapse collapse navbar-responsive-collapse">
|
||
<ul class="nav navbar-nav navbar-right">
|
||
<li class="dropdown">
|
||
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
|
||
<ul class="dropdown-menu">
|
||
<!-- navigation toc: --> <li><a href="._week41-bs001.html#___sec0" style="font-size: 80%;">Plan for week 41</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs002.html#___sec1" style="font-size: 80%;">Setting up the Back propagation algorithm</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs003.html#___sec2" style="font-size: 80%;">Setting up a Multi-layer perceptron model for classification</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs004.html#___sec3" style="font-size: 80%;">Defining the cost function</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs005.html#___sec4" style="font-size: 80%;">Example: binary classification problem</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs006.html#___sec5" style="font-size: 80%;">The Softmax function</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs007.html#___sec6" style="font-size: 80%;">Developing a code for doing neural networks with back propagation</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs008.html#___sec7" style="font-size: 80%;">Collect and pre-process data</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs009.html#___sec8" style="font-size: 80%;">Train and test datasets</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs010.html#___sec9" style="font-size: 80%;">Define model and architecture</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs011.html#___sec10" style="font-size: 80%;">Layers</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs012.html#___sec11" style="font-size: 80%;">Weights and biases</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs013.html#___sec12" style="font-size: 80%;">Feed-forward pass</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs014.html#___sec13" style="font-size: 80%;">Matrix multiplications</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs015.html#___sec14" style="font-size: 80%;">Choose cost function and optimizer</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs016.html#___sec15" style="font-size: 80%;">Optimizing the cost function</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs017.html#___sec16" style="font-size: 80%;">Regularization</a></li>
|
||
<!-- navigation toc: --> <li><a href="#___sec17" style="font-size: 80%;">Matrix multiplication</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs019.html#___sec18" style="font-size: 80%;">Improving performance</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs020.html#___sec19" style="font-size: 80%;">Full object-oriented implementation</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs021.html#___sec20" style="font-size: 80%;">Evaluate model performance on test data</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs022.html#___sec21" style="font-size: 80%;">Adjust hyperparameters</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs023.html#___sec22" style="font-size: 80%;">Visualization</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs024.html#___sec23" style="font-size: 80%;">scikit-learn implementation</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs025.html#___sec24" style="font-size: 80%;">Visualization</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs026.html#___sec25" style="font-size: 80%;">Building neural networks in Tensorflow and Keras</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs027.html#___sec26" style="font-size: 80%;">Tensorflow</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs028.html#___sec27" style="font-size: 80%;">Using Keras</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs029.html#___sec28" style="font-size: 80%;">Collect and pre-process data</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs030.html#___sec29" style="font-size: 80%;">The Breast Cancer Data, now with Keras</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs031.html#___sec30" style="font-size: 80%;">Fine-tuning neural network hyperparameters</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs032.html#___sec31" style="font-size: 80%;">Hidden layers</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs033.html#___sec32" style="font-size: 80%;">Which activation function should I use?</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs034.html#___sec33" style="font-size: 80%;">Is the Logistic activation function (Sigmoid) our choice?</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs035.html#___sec34" style="font-size: 80%;">The derivative of the Logistic funtion</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs036.html#___sec35" style="font-size: 80%;">The RELU function family</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs037.html#___sec36" style="font-size: 80%;">Which activation function should we use?</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs038.html#___sec37" style="font-size: 80%;">More on activation functions, output layers</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs039.html#___sec38" style="font-size: 80%;">Batch Normalization</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs040.html#___sec39" style="font-size: 80%;">Dropout</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs041.html#___sec40" style="font-size: 80%;">Gradient Clipping</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs042.html#___sec41" style="font-size: 80%;">A very nice website on Neural Networks</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs043.html#___sec42" style="font-size: 80%;">A top-down perspective on Neural networks</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs044.html#___sec43" style="font-size: 80%;">Limitations of supervised learning with deep networks</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs045.html#___sec44" style="font-size: 80%;">Convolutional Neural Networks (recognizing images)</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs046.html#___sec45" style="font-size: 80%;">Regular NNs don’t scale well to full images</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs047.html#___sec46" style="font-size: 80%;">3D volumes of neurons</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs048.html#___sec47" style="font-size: 80%;">Layers used to build CNNs</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs049.html#___sec48" style="font-size: 80%;">Transforming images</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs050.html#___sec49" style="font-size: 80%;">CNNs in brief</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs051.html#___sec50" style="font-size: 80%;">CNNs in more detail, building convolutional neural networks in Tensorflow and Keras</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs052.html#___sec51" style="font-size: 80%;">Setting it up</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs053.html#___sec52" style="font-size: 80%;">The MNIST dataset again</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs054.html#___sec53" style="font-size: 80%;">Strong correlations</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs055.html#___sec54" style="font-size: 80%;">Layers of a CNN</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs056.html#___sec55" style="font-size: 80%;">Systematic reduction</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs057.html#___sec56" style="font-size: 80%;">Prerequisites: Collect and pre-process data</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs058.html#___sec57" style="font-size: 80%;">Importing Keras and Tensorflow</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs059.html#___sec58" style="font-size: 80%;">Running with Keras</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs060.html#___sec59" style="font-size: 80%;">Final part</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs061.html#___sec60" style="font-size: 80%;">Final visualization</a></li>
|
||
<!-- navigation toc: --> <li><a href="._week41-bs062.html#___sec61" style="font-size: 80%;">Fun links</a></li>
|
||
|
||
</ul>
|
||
</li>
|
||
</ul>
|
||
</div>
|
||
</div>
|
||
</div> <!-- end of navigation bar -->
|
||
|
||
<div class="container">
|
||
|
||
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
||
|
||
<a name="part0018"></a>
|
||
<!-- !split -->
|
||
|
||
<h2 id="___sec17" class="anchor">Matrix multiplication </h2>
|
||
|
||
<p>
|
||
To more efficently train our network these equations are implemented using matrix operations.
|
||
The error in the output layer is calculated simply as, with \( \hat{t} \) being our targets,
|
||
|
||
$$ \delta_L = \hat{t} - \hat{y} = (n_{inputs}, n_{categories}) .$$
|
||
|
||
<p>
|
||
The gradient for the output weights is calculated as
|
||
|
||
$$ \nabla W_{L} = \hat{a}^T \delta_L = (n_{hidden}, n_{categories}) ,$$
|
||
|
||
<p>
|
||
where \( \hat{a} = (n_{inputs}, n_{hidden}) \). This simply means that we are summing up the gradients for each input.
|
||
Since we are going backwards we have to transpose the activation matrix.
|
||
|
||
<p>
|
||
The gradient with respect to the output bias is then
|
||
|
||
$$ \nabla \hat{b}_{L} = \sum_{i=1}^{n_{inputs}} \delta_L = (n_{categories}) .$$
|
||
|
||
<p>
|
||
The error in the hidden layer is
|
||
|
||
$$ \Delta_h = \delta_L W_{L}^T \circ f'(z_{h}) = \delta_L W_{L}^T \circ a_{h} \circ (1 - a_{h}) = (n_{inputs}, n_{hidden}) ,$$
|
||
|
||
<p>
|
||
where \( f'(a_{h}) \) is the derivative of the activation in the hidden layer. The matrix products mean
|
||
that we are summing up the products for each neuron in the output layer. The symbol \( \circ \) denotes
|
||
the <em>Hadamard product</em>, meaning element-wise multiplication.
|
||
|
||
<p>
|
||
This again gives us the gradients in the hidden layer:
|
||
|
||
$$ \nabla W_{h} = X^T \delta_h = (n_{features}, n_{hidden}) ,$$
|
||
|
||
$$ \nabla b_{h} = \sum_{i=1}^{n_{inputs}} \delta_h = (n_{hidden}) .$$
|
||
|
||
<p>
|
||
|
||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%"><span></span><span style="color: #408080; font-style: italic"># to categorical turns our integer vector into a onehot representation</span>
|
||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.metrics</span> <span style="color: #008000; font-weight: bold">import</span> accuracy_score
|
||
|
||
<span style="color: #408080; font-style: italic"># one-hot in numpy</span>
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">to_categorical_numpy</span>(integer_vector):
|
||
n_inputs <span style="color: #666666">=</span> <span style="color: #008000">len</span>(integer_vector)
|
||
n_categories <span style="color: #666666">=</span> np<span style="color: #666666">.</span>max(integer_vector) <span style="color: #666666">+</span> <span style="color: #666666">1</span>
|
||
onehot_vector <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n_inputs, n_categories))
|
||
onehot_vector[<span style="color: #008000">range</span>(n_inputs), integer_vector] <span style="color: #666666">=</span> <span style="color: #666666">1</span>
|
||
|
||
<span style="color: #008000; font-weight: bold">return</span> onehot_vector
|
||
|
||
<span style="color: #408080; font-style: italic">#Y_train_onehot, Y_test_onehot = to_categorical(Y_train), to_categorical(Y_test)</span>
|
||
Y_train_onehot, Y_test_onehot <span style="color: #666666">=</span> to_categorical_numpy(Y_train), to_categorical_numpy(Y_test)
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">feed_forward_train</span>(X):
|
||
<span style="color: #408080; font-style: italic"># weighted sum of inputs to the hidden layer</span>
|
||
z_h <span style="color: #666666">=</span> np<span style="color: #666666">.</span>matmul(X, hidden_weights) <span style="color: #666666">+</span> hidden_bias
|
||
<span style="color: #408080; font-style: italic"># activation in the hidden layer</span>
|
||
a_h <span style="color: #666666">=</span> sigmoid(z_h)
|
||
|
||
<span style="color: #408080; font-style: italic"># weighted sum of inputs to the output layer</span>
|
||
z_o <span style="color: #666666">=</span> np<span style="color: #666666">.</span>matmul(a_h, output_weights) <span style="color: #666666">+</span> output_bias
|
||
<span style="color: #408080; font-style: italic"># softmax output</span>
|
||
<span style="color: #408080; font-style: italic"># axis 0 holds each input and axis 1 the probabilities of each category</span>
|
||
exp_term <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(z_o)
|
||
probabilities <span style="color: #666666">=</span> exp_term <span style="color: #666666">/</span> np<span style="color: #666666">.</span>sum(exp_term, axis<span style="color: #666666">=1</span>, keepdims<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)
|
||
|
||
<span style="color: #408080; font-style: italic"># for backpropagation need activations in hidden and output layers</span>
|
||
<span style="color: #008000; font-weight: bold">return</span> a_h, probabilities
|
||
|
||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">backpropagation</span>(X, Y):
|
||
a_h, probabilities <span style="color: #666666">=</span> feed_forward_train(X)
|
||
|
||
<span style="color: #408080; font-style: italic"># error in the output layer</span>
|
||
error_output <span style="color: #666666">=</span> probabilities <span style="color: #666666">-</span> Y
|
||
<span style="color: #408080; font-style: italic"># error in the hidden layer</span>
|
||
error_hidden <span style="color: #666666">=</span> np<span style="color: #666666">.</span>matmul(error_output, output_weights<span style="color: #666666">.</span>T) <span style="color: #666666">*</span> a_h <span style="color: #666666">*</span> (<span style="color: #666666">1</span> <span style="color: #666666">-</span> a_h)
|
||
|
||
<span style="color: #408080; font-style: italic"># gradients for the output layer</span>
|
||
output_weights_gradient <span style="color: #666666">=</span> np<span style="color: #666666">.</span>matmul(a_h<span style="color: #666666">.</span>T, error_output)
|
||
output_bias_gradient <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(error_output, axis<span style="color: #666666">=0</span>)
|
||
|
||
<span style="color: #408080; font-style: italic"># gradient for the hidden layer</span>
|
||
hidden_weights_gradient <span style="color: #666666">=</span> np<span style="color: #666666">.</span>matmul(X<span style="color: #666666">.</span>T, error_hidden)
|
||
hidden_bias_gradient <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(error_hidden, axis<span style="color: #666666">=0</span>)
|
||
|
||
<span style="color: #008000; font-weight: bold">return</span> output_weights_gradient, output_bias_gradient, hidden_weights_gradient, hidden_bias_gradient
|
||
|
||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Old accuracy on training data: "</span> <span style="color: #666666">+</span> <span style="color: #008000">str</span>(accuracy_score(predict(X_train), Y_train)))
|
||
|
||
eta <span style="color: #666666">=</span> <span style="color: #666666">0.01</span>
|
||
lmbd <span style="color: #666666">=</span> <span style="color: #666666">0.01</span>
|
||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1000</span>):
|
||
<span style="color: #408080; font-style: italic"># calculate gradients</span>
|
||
dWo, dBo, dWh, dBh <span style="color: #666666">=</span> backpropagation(X_train, Y_train_onehot)
|
||
|
||
<span style="color: #408080; font-style: italic"># regularization term gradients</span>
|
||
dWo <span style="color: #666666">+=</span> lmbd <span style="color: #666666">*</span> output_weights
|
||
dWh <span style="color: #666666">+=</span> lmbd <span style="color: #666666">*</span> hidden_weights
|
||
|
||
<span style="color: #408080; font-style: italic"># update weights and biases</span>
|
||
output_weights <span style="color: #666666">-=</span> eta <span style="color: #666666">*</span> dWo
|
||
output_bias <span style="color: #666666">-=</span> eta <span style="color: #666666">*</span> dBo
|
||
hidden_weights <span style="color: #666666">-=</span> eta <span style="color: #666666">*</span> dWh
|
||
hidden_bias <span style="color: #666666">-=</span> eta <span style="color: #666666">*</span> dBh
|
||
|
||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"New accuracy on training data: "</span> <span style="color: #666666">+</span> <span style="color: #008000">str</span>(accuracy_score(predict(X_train), Y_train)))
|
||
</pre></div>
|
||
<p>
|
||
<p>
|
||
<!-- navigation buttons at the bottom of the page -->
|
||
<ul class="pagination">
|
||
<li><a href="._week41-bs017.html">«</a></li>
|
||
<li><a href="._week41-bs000.html">1</a></li>
|
||
<li><a href="">...</a></li>
|
||
<li><a href="._week41-bs010.html">11</a></li>
|
||
<li><a href="._week41-bs011.html">12</a></li>
|
||
<li><a href="._week41-bs012.html">13</a></li>
|
||
<li><a href="._week41-bs013.html">14</a></li>
|
||
<li><a href="._week41-bs014.html">15</a></li>
|
||
<li><a href="._week41-bs015.html">16</a></li>
|
||
<li><a href="._week41-bs016.html">17</a></li>
|
||
<li><a href="._week41-bs017.html">18</a></li>
|
||
<li class="active"><a href="._week41-bs018.html">19</a></li>
|
||
<li><a href="._week41-bs019.html">20</a></li>
|
||
<li><a href="._week41-bs020.html">21</a></li>
|
||
<li><a href="._week41-bs021.html">22</a></li>
|
||
<li><a href="._week41-bs022.html">23</a></li>
|
||
<li><a href="._week41-bs023.html">24</a></li>
|
||
<li><a href="._week41-bs024.html">25</a></li>
|
||
<li><a href="._week41-bs025.html">26</a></li>
|
||
<li><a href="._week41-bs026.html">27</a></li>
|
||
<li><a href="._week41-bs027.html">28</a></li>
|
||
<li><a href="">...</a></li>
|
||
<li><a href="._week41-bs062.html">63</a></li>
|
||
<li><a href="._week41-bs019.html">»</a></li>
|
||
</ul>
|
||
<!-- ------------------- end of main content --------------- -->
|
||
|
||
</div> <!-- end container -->
|
||
<!-- include javascript, jQuery *first* -->
|
||
<script src="https://ajax.googleapis.com/ajax/libs/jquery/1.10.2/jquery.min.js"></script>
|
||
<script src="https://netdna.bootstrapcdn.com/bootstrap/3.0.0/js/bootstrap.min.js"></script>
|
||
|
||
<!-- Bootstrap footer
|
||
<footer>
|
||
<a href="http://..."><img width="250" align=right src="http://..."></a>
|
||
</footer>
|
||
-->
|
||
|
||
|
||
<center style="font-size:80%">
|
||
<!-- copyright only on the titlepage -->
|
||
</center>
|
||
|
||
|
||
</body>
|
||
</html>
|
||
|
||
|