From 2bb9615d1cba4572d9dc062e5c29bc60ecc2806f Mon Sep 17 00:00:00 2001
From: Morten Hjorth-Jensen
Date: Mon, 8 Sep 2025 07:54:02 +0200
Subject: [PATCH] update week 37
---
doc/pub/week37/html/._week37-bs000.html | 160 ++--
doc/pub/week37/html/._week37-bs001.html | 160 ++--
doc/pub/week37/html/._week37-bs002.html | 160 ++--
doc/pub/week37/html/._week37-bs003.html | 160 ++--
doc/pub/week37/html/._week37-bs004.html | 160 ++--
doc/pub/week37/html/._week37-bs005.html | 160 ++--
doc/pub/week37/html/._week37-bs006.html | 160 ++--
doc/pub/week37/html/._week37-bs007.html | 160 ++--
doc/pub/week37/html/._week37-bs008.html | 160 ++--
doc/pub/week37/html/._week37-bs009.html | 160 ++--
doc/pub/week37/html/._week37-bs010.html | 160 ++--
doc/pub/week37/html/._week37-bs011.html | 160 ++--
doc/pub/week37/html/._week37-bs012.html | 160 ++--
doc/pub/week37/html/._week37-bs013.html | 160 ++--
doc/pub/week37/html/._week37-bs014.html | 160 ++--
doc/pub/week37/html/._week37-bs015.html | 160 ++--
doc/pub/week37/html/._week37-bs016.html | 160 ++--
doc/pub/week37/html/._week37-bs017.html | 160 ++--
doc/pub/week37/html/._week37-bs018.html | 160 ++--
doc/pub/week37/html/._week37-bs019.html | 160 ++--
doc/pub/week37/html/._week37-bs020.html | 160 ++--
doc/pub/week37/html/._week37-bs021.html | 160 ++--
doc/pub/week37/html/._week37-bs022.html | 160 ++--
doc/pub/week37/html/._week37-bs023.html | 160 ++--
doc/pub/week37/html/._week37-bs024.html | 160 ++--
doc/pub/week37/html/._week37-bs025.html | 160 ++--
doc/pub/week37/html/._week37-bs026.html | 160 ++--
doc/pub/week37/html/._week37-bs027.html | 160 ++--
doc/pub/week37/html/._week37-bs028.html | 160 ++--
doc/pub/week37/html/._week37-bs029.html | 160 ++--
doc/pub/week37/html/._week37-bs030.html | 160 ++--
doc/pub/week37/html/._week37-bs031.html | 160 ++--
doc/pub/week37/html/._week37-bs032.html | 160 ++--
doc/pub/week37/html/._week37-bs033.html | 281 ++++---
doc/pub/week37/html/._week37-bs034.html | 224 ++++--
doc/pub/week37/html/._week37-bs035.html | 244 +++++--
doc/pub/week37/html/._week37-bs036.html | 202 +++---
doc/pub/week37/html/._week37-bs037.html | 193 ++---
doc/pub/week37/html/._week37-bs038.html | 183 +++--
doc/pub/week37/html/._week37-bs039.html | 197 ++---
doc/pub/week37/html/._week37-bs040.html | 187 +++--
doc/pub/week37/html/._week37-bs041.html | 181 +++--
doc/pub/week37/html/._week37-bs042.html | 186 +++--
doc/pub/week37/html/._week37-bs043.html | 192 ++---
doc/pub/week37/html/._week37-bs044.html | 180 +++--
doc/pub/week37/html/._week37-bs045.html | 180 +++--
doc/pub/week37/html/._week37-bs046.html | 196 +++--
doc/pub/week37/html/._week37-bs047.html | 184 +++--
doc/pub/week37/html/._week37-bs048.html | 180 +++--
doc/pub/week37/html/._week37-bs049.html | 176 +++--
doc/pub/week37/html/week37-bs.html | 160 ++--
doc/pub/week37/html/week37-reveal.html | 247 +++++++
doc/pub/week37/html/week37-solarized.html | 254 +++++++
doc/pub/week37/html/week37.html | 254 +++++++
doc/pub/week37/ipynb/ipynb-week37-src.tar.gz | Bin 492464 -> 492464 bytes
doc/pub/week37/ipynb/week37.ipynb | 726 ++++++++++++++-----
doc/src/week37/Latexfiles/sgd.txt | 246 +++++++
doc/src/week37/week37.do.txt | 246 +++++++
58 files changed, 7013 insertions(+), 3766 deletions(-)
create mode 100644 doc/src/week37/Latexfiles/sgd.txt
diff --git a/doc/pub/week37/html/._week37-bs000.html b/doc/pub/week37/html/._week37-bs000.html
index 6ff6b267b..5cb7bff40 100644
--- a/doc/pub/week37/html/._week37-bs000.html
+++ b/doc/pub/week37/html/._week37-bs000.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -388,7 +416,7 @@ MathJax.Hub.Config({
9
10
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs001.html b/doc/pub/week37/html/._week37-bs001.html
index 19c0b741c..d3d5205cb 100644
--- a/doc/pub/week37/html/._week37-bs001.html
+++ b/doc/pub/week37/html/._week37-bs001.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -379,7 +407,7 @@ MathJax.Hub.Config({
10
11
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs002.html b/doc/pub/week37/html/._week37-bs002.html
index 08e3c6566..1ea756d6e 100644
--- a/doc/pub/week37/html/._week37-bs002.html
+++ b/doc/pub/week37/html/._week37-bs002.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -377,7 +405,7 @@ MathJax.Hub.Config({
11
12
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs003.html b/doc/pub/week37/html/._week37-bs003.html
index f35594c2d..1efdbacc7 100644
--- a/doc/pub/week37/html/._week37-bs003.html
+++ b/doc/pub/week37/html/._week37-bs003.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -366,7 +394,7 @@ MathJax.Hub.Config({
12
13
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs004.html b/doc/pub/week37/html/._week37-bs004.html
index b8669234a..b160fe6fe 100644
--- a/doc/pub/week37/html/._week37-bs004.html
+++ b/doc/pub/week37/html/._week37-bs004.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -418,7 +446,7 @@ $$
13
14
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs005.html b/doc/pub/week37/html/._week37-bs005.html
index 6b77645b5..89cb2368c 100644
--- a/doc/pub/week37/html/._week37-bs005.html
+++ b/doc/pub/week37/html/._week37-bs005.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -386,7 +414,7 @@ $$
14
15
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs006.html b/doc/pub/week37/html/._week37-bs006.html
index 26d36ef6b..623def450 100644
--- a/doc/pub/week37/html/._week37-bs006.html
+++ b/doc/pub/week37/html/._week37-bs006.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -378,7 +406,7 @@ $$
15
16
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs007.html b/doc/pub/week37/html/._week37-bs007.html
index d758b7c4d..9765369b2 100644
--- a/doc/pub/week37/html/._week37-bs007.html
+++ b/doc/pub/week37/html/._week37-bs007.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -379,7 +407,7 @@ $$
16
17
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs008.html b/doc/pub/week37/html/._week37-bs008.html
index 809be722a..80e607a76 100644
--- a/doc/pub/week37/html/._week37-bs008.html
+++ b/doc/pub/week37/html/._week37-bs008.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -385,7 +413,7 @@ when \( ||\nabla_\theta C(\theta_k) || \leq \epsilon = 10^{-8} \). Note that
17
18
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs009.html b/doc/pub/week37/html/._week37-bs009.html
index 35ca99d29..696a4305c 100644
--- a/doc/pub/week37/html/._week37-bs009.html
+++ b/doc/pub/week37/html/._week37-bs009.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -441,7 +469,7 @@ plt.show()
18
19
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs010.html b/doc/pub/week37/html/._week37-bs010.html
index 4cad87a15..269eb7e48 100644
--- a/doc/pub/week37/html/._week37-bs010.html
+++ b/doc/pub/week37/html/._week37-bs010.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -391,7 +419,7 @@ $$
19
20
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs011.html b/doc/pub/week37/html/._week37-bs011.html
index cc4dd0186..0e2776240 100644
--- a/doc/pub/week37/html/._week37-bs011.html
+++ b/doc/pub/week37/html/._week37-bs011.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -387,7 +415,7 @@ minimum of this function.
20
21
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs012.html b/doc/pub/week37/html/._week37-bs012.html
index 207bc4790..89dbfd4cb 100644
--- a/doc/pub/week37/html/._week37-bs012.html
+++ b/doc/pub/week37/html/._week37-bs012.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -445,7 +473,7 @@ plt.show()
21
22
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs013.html b/doc/pub/week37/html/._week37-bs013.html
index fef2d57ec..31dcd4508 100644
--- a/doc/pub/week37/html/._week37-bs013.html
+++ b/doc/pub/week37/html/._week37-bs013.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -381,7 +409,7 @@ MathJax.Hub.Config({
22
23
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs014.html b/doc/pub/week37/html/._week37-bs014.html
index b202ece41..fd2ab0e0b 100644
--- a/doc/pub/week37/html/._week37-bs014.html
+++ b/doc/pub/week37/html/._week37-bs014.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -379,7 +407,7 @@ For the mathematical details, see whiteboad notes from lecture on September 8, 2
23
24
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs015.html b/doc/pub/week37/html/._week37-bs015.html
index ad33ddee4..f11f51071 100644
--- a/doc/pub/week37/html/._week37-bs015.html
+++ b/doc/pub/week37/html/._week37-bs015.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -451,7 +479,7 @@ pyplot.show()
24
25
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs016.html b/doc/pub/week37/html/._week37-bs016.html
index 70cbd2894..13fcc4f3c 100644
--- a/doc/pub/week37/html/._week37-bs016.html
+++ b/doc/pub/week37/html/._week37-bs016.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -459,7 +487,7 @@ pyplot.show()
25
26
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs017.html b/doc/pub/week37/html/._week37-bs017.html
index 93765488e..443f4a91c 100644
--- a/doc/pub/week37/html/._week37-bs017.html
+++ b/doc/pub/week37/html/._week37-bs017.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -381,7 +409,7 @@ MathJax.Hub.Config({
26
27
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs018.html b/doc/pub/week37/html/._week37-bs018.html
index d2a6cef4f..d76d509b2 100644
--- a/doc/pub/week37/html/._week37-bs018.html
+++ b/doc/pub/week37/html/._week37-bs018.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -385,7 +413,7 @@ perform a parameter update.
27
28
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs019.html b/doc/pub/week37/html/._week37-bs019.html
index 9abb89c6a..82729b77d 100644
--- a/doc/pub/week37/html/._week37-bs019.html
+++ b/doc/pub/week37/html/._week37-bs019.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -378,7 +406,7 @@ MathJax.Hub.Config({
28
29
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs020.html b/doc/pub/week37/html/._week37-bs020.html
index 325b26df1..48a11cdf1 100644
--- a/doc/pub/week37/html/._week37-bs020.html
+++ b/doc/pub/week37/html/._week37-bs020.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -377,7 +405,7 @@ MathJax.Hub.Config({
29
30
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs021.html b/doc/pub/week37/html/._week37-bs021.html
index 91daef69f..96a20cf2a 100644
--- a/doc/pub/week37/html/._week37-bs021.html
+++ b/doc/pub/week37/html/._week37-bs021.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -394,7 +422,7 @@ such as momentum or adaptive learning rates
30
31
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs022.html b/doc/pub/week37/html/._week37-bs022.html
index 66c8d2fc3..2ea8fe151 100644
--- a/doc/pub/week37/html/._week37-bs022.html
+++ b/doc/pub/week37/html/._week37-bs022.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -397,7 +425,7 @@ sized in powers of 2.
31
32
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs023.html b/doc/pub/week37/html/._week37-bs023.html
index 4af9644fd..9ec237128 100644
--- a/doc/pub/week37/html/._week37-bs023.html
+++ b/doc/pub/week37/html/._week37-bs023.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -387,7 +415,7 @@ $$
32
33
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs024.html b/doc/pub/week37/html/._week37-bs024.html
index 97d263e47..fb7c448b1 100644
--- a/doc/pub/week37/html/._week37-bs024.html
+++ b/doc/pub/week37/html/._week37-bs024.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -388,7 +416,7 @@ minibatches. We denote these minibatches by \( B_k \) where
33
34
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs025.html b/doc/pub/week37/html/._week37-bs025.html
index 216436bb2..86fad8686 100644
--- a/doc/pub/week37/html/._week37-bs025.html
+++ b/doc/pub/week37/html/._week37-bs025.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -394,7 +422,7 @@ $$
34
35
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs026.html b/doc/pub/week37/html/._week37-bs026.html
index f068d2f81..1de6b2d78 100644
--- a/doc/pub/week37/html/._week37-bs026.html
+++ b/doc/pub/week37/html/._week37-bs026.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -386,7 +414,7 @@ the number of minibatches, as exemplified in the code below.
35
36
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs027.html b/doc/pub/week37/html/._week37-bs027.html
index 4b2546ff0..c6c9aca48 100644
--- a/doc/pub/week37/html/._week37-bs027.html
+++ b/doc/pub/week37/html/._week37-bs027.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -418,7 +446,7 @@ all \( n \) datapoints.
36
37
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs028.html b/doc/pub/week37/html/._week37-bs028.html
index 5b61000ee..03abfb9aa 100644
--- a/doc/pub/week37/html/._week37-bs028.html
+++ b/doc/pub/week37/html/._week37-bs028.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -385,7 +413,7 @@ gave the lowest value.
37
38
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs029.html b/doc/pub/week37/html/._week37-bs029.html
index 79df1c3a0..1049e2326 100644
--- a/doc/pub/week37/html/._week37-bs029.html
+++ b/doc/pub/week37/html/._week37-bs029.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -384,7 +412,7 @@ for a discussion of different scaling functions for the learning rate.
38
39
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs030.html b/doc/pub/week37/html/._week37-bs030.html
index 4bd362daa..aec105dd2 100644
--- a/doc/pub/week37/html/._week37-bs030.html
+++ b/doc/pub/week37/html/._week37-bs030.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -429,7 +457,7 @@ j = 0
39
40
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs031.html b/doc/pub/week37/html/._week37-bs031.html
index fa59b710f..2c6d23dea 100644
--- a/doc/pub/week37/html/._week37-bs031.html
+++ b/doc/pub/week37/html/._week37-bs031.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -463,7 +491,7 @@ plt.show()
40
41
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs032.html b/doc/pub/week37/html/._week37-bs032.html
index e5947ea97..5bcaa9a17 100644
--- a/doc/pub/week37/html/._week37-bs032.html
+++ b/doc/pub/week37/html/._week37-bs032.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -379,7 +407,7 @@ useful.
41
42
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs033.html b/doc/pub/week37/html/._week37-bs033.html
index c80c1882d..220f12a1e 100644
--- a/doc/pub/week37/html/._week37-bs033.html
+++ b/doc/pub/week37/html/._week37-bs033.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,29 +374,110 @@ MathJax.Hub.Config({
-Second moment of the gradient
+SGD vs Full-Batch GD: Convergence Speed and Memory Comparison
+Theoretical Convergence Speed and convex optimization
-In stochastic gradient descent, with and without momentum, we still
-have to specify a schedule for tuning the learning rates \( \eta_t \)
-as a function of time. As discussed in the context of Newton's
-method, this presents a number of dilemmas. The learning rate is
-limited by the steepest direction which can change depending on the
-current position in the landscape. To circumvent this problem, ideally
-our algorithm would keep track of curvature and take large steps in
-shallow, flat directions and small steps in steep, narrow directions.
-Second-order methods accomplish this by calculating or approximating
-the Hessian and normalizing the learning rate by the
-curvature. However, this is very computationally expensive for
-extremely large models. Ideally, we would like to be able to
-adaptively change the step size to match the landscape without paying
-the steep computational price of calculating or approximating
-Hessians.
+
Consider minimizing an empirical cost function
+$$
+C(\theta) =\frac{1}{N}\sum_{i=1}^N l_i(\theta),
+$$
+
+where each \( l_i(\theta) \) is a
+differentiable loss term. Gradient Descent (GD) updates parameters
+using the full gradient \( \nabla C(\theta) \), while Stochastic Gradient
+Descent (SGD) uses a single sample (or mini-batch) gradient \( \nabla
+l_i(\theta) \) selected at random. In equation form, one GD step is:
-During the last decade a number of methods have been introduced that accomplish
-this by tracking not only the gradient, but also the second moment of
-the gradient. These methods include AdaGrad, AdaDelta, Root Mean Squared Propagation (RMS-Prop), and
-ADAM.
+$$
+\theta_{t+1} = \theta_t-\eta \nabla C(\theta_t) =\theta_t -\eta \frac{1}{N}\sum_{i=1}^N \nabla l_i(\theta_t),
+$$
+
+
whereas one SGD step is:
+
+$$
+\theta_{t+1} = \theta_t -\eta \nabla l_{i_t}(\theta_t),
+$$
+
+with \( i_t \) randomly chosen. On smooth convex problems, GD and SGD both
+converge to the global minimum, but their rates differ. GD can take
+larger, more stable steps since it uses the exact gradient, achieving
+an error that decreases on the order of \( O(1/t) \) per iteration for
+convex objectives (and even exponentially fast for strongly convex
+cases). In contrast, plain SGD has more variance in each step, leading
+to sublinear convergence in expectation – typically \( O(1/\sqrt{t}) \)
+for general convex objectives (\thetaith appropriate diminishing step
+sizes) . Intuitively, GD’s trajectory is smoother and more
+predictable, while SGD’s path oscillates due to noise but costs far
+less per iteration, enabling many more updates in the same time.
+
+Strongly Convex Case
+
+If \( C(\theta) \) is strongly convex and \( L \)-smooth (so GD enjoys linear
+convergence), the gap \( C(\theta_t)-C(\theta^*) \) for GD shrinks as
+
+$$
+C(\theta_t) - C(\theta^* ) \le \Big(1 - \frac{\mu}{L}\Big)^t [C(\theta_0)-C(\theta^*)],
+$$
+
+a geometric (linear) convergence per iteration . Achieving an
+\( \epsilon \)-accurate solution thus takes on the order of
+\( \log(1/\epsilon) \) iterations for GD. However, each GD iteration costs
+\( O(N) \) gradient evaluations. SGD cannot exploit strong convexity to
+obtain a linear rate – instead, with a properly decaying step size
+(e.g. \( \eta_t = \frac{1}{\mu t} \)) or iterate averaging, SGD attains an
+\( O(1/t) \) convergence rate in expectation . For example, one result
+of Moulines and Bach 2011, see https://papers.nips.cc/paper_files/paper/2011/hash/40008b9a5380fcacce3976bf7c08af5b-Abstract.html shows that with \( \eta_t = \Theta(1/t) \),
+
+$$
+\mathbb{E}[C(\theta_t) - C(\theta^*)] = O(1/t),
+$$
+
+for strongly convex, smooth \( F \) . This \( 1/t \) rate is slower per
+iteration than GD’s exponential decay, but each SGD iteration is \( N \)
+times cheaper. In fact, to reach error \( \epsilon \), plain SGD needs on
+the order of \( T=O(1/\epsilon) \) iterations (sub-linear convergence),
+while GD needs \( O(\log(1/\epsilon)) \) iterations. When accounting for
+cost-per-iteration, GD requires \( O(N \log(1/\epsilon)) \) total gradient
+computations versus SGD’s \( O(1/\epsilon) \) single-sample
+computations. In large-scale regimes (huge \( N \)), SGD can be
+faster in wall-clock time because \( N \log(1/\epsilon) \) may far exceed
+\( 1/\epsilon \) for reasonable accuracy levels. In other words,
+with millions of data points, one epoch of GD (one full gradient) is
+extremely costly, whereas SGD can make \( N \) cheap updates in the time
+GD makes one – often yielding a good solution faster in practice, even
+though SGD’s asymptotic error decays more slowly. As one lecture
+succinctly puts it: “SGD can be super effective in terms of iteration
+cost and memory, but SGD is slow to converge and can’t adapt to strong
+convexity” . Thus, the break-even point depends on \( N \) and the desired
+accuracy: for moderate accuracy on very large \( N \), SGD’s cheaper
+updates win; for extremely high precision (very small \( \epsilon \)) on a
+modest \( N \), GD’s fast convergence per step can be advantageous.
+
+Non-Convex Problems
+
+In non-convex optimization (e.g. deep neural networks), neither GD nor
+SGD guarantees global minima, but SGD often displays faster progress
+in finding useful minima. Theoretical results here are weaker, usually
+showing convergence to a stationary point \( \theta \) (\( |\nabla C| \) is
+small) in expectation. For example, GD might require \( O(1/\epsilon^2) \)
+iterations to ensure \( |\nabla C(\theta)| < \epsilon \), and SGD typically has
+similar polynomial complexity (often worse due to gradient
+noise). However, a noteworthy difference is that SGD’s stochasticity
+can help escape saddle points or poor local minima. Random gradient
+fluctuations act like implicit noise, helping the iterate “jump” out
+of flat saddle regions where full-batch GD could stagnate . In fact,
+research has shown that adding noise to GD can guarantee escaping
+saddle points in polynomial time, and the inherent noise in SGD often
+serves this role. Empirically, this means SGD can sometimes find a
+lower loss basin faster, whereas full-batch GD might get “stuck” near
+saddle points or need a very small learning rate to navigate complex
+error surfaces . Overall, in modern high-dimensional machine learning,
+SGD (or mini-batch SGD) is the workhorse for large non-convex problems
+because it converges to good solutions much faster in practice,
+despite the lack of a linear convergence guarantee. Full-batch GD is
+rarely used on large neural networks, as it would require tiny steps
+to avoid divergence and is extremely slow per iteration .
@@ -396,7 +505,7 @@ the gradient. These methods include AdaGrad, AdaDelta, Root Mean Squared Propaga
42
43
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs034.html b/doc/pub/week37/html/._week37-bs034.html
index ed4a6e925..932fa4f2d 100644
--- a/doc/pub/week37/html/._week37-bs034.html
+++ b/doc/pub/week37/html/._week37-bs034.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,20 +374,58 @@ MathJax.Hub.Config({
-Challenge: Choosing a Fixed Learning Rate
-A fixed \( \eta \) is hard to get right:
-
-- If \( \eta \) is too large, the updates can overshoot the minimum, causing oscillations or divergence
-- If \( \eta \) is too small, convergence is very slow (many iterations to make progress)
-
-In practice, one often uses trial-and-error or schedules (decaying \( \eta \) over time) to find a workable balance.
-For a function with steep directions and flat directions, a single global \( \eta \) may be inappropriate:
+
Memory Usage and Scalability
+
+A major advantage of SGD is its memory efficiency in handling large
+datasets. Full-batch GD requires access to the entire training set for
+each iteration, which often means the whole dataset (or a large
+subset) must reside in memory to compute \( \nabla C(\theta) \) . This results
+in memory usage that scales linearly with the dataset size \( N \). For
+instance, if each training sample is large (e.g. high-dimensional
+features), computing a full gradient may require storing a substantial
+portion of the data or all intermediate gradients until they are
+aggregated. In contrast, SGD needs only a single (or a small
+mini-batch of) training example(s) in memory at any time . The
+algorithm processes one sample (or mini-batch) at a time and
+immediately updates the model, discarding that sample before moving to
+the next. This streaming approach means that memory footprint is
+essentially independent of \( N \) (apart from storing the model
+parameters themselves). As one source notes, gradient descent
+“requires more memory than SGD” because it “must store the entire
+dataset for each iteration,” whereas SGD “only needs to store the
+current training example” . In practical terms, if you have a dataset
+of size, say, 1 million examples, full-batch GD would need memory for
+all million every step, while SGD could be implemented to load just
+one example at a time – a crucial benefit if data are too large to fit
+in RAM or GPU memory. This scalability makes SGD suitable for
+large-scale learning: as long as you can stream data from disk, SGD
+can handle arbitrarily large datasets with fixed memory. In fact, SGD
+“does not need to remember which examples were visited” in the past,
+allowing it to run in an online fashion on infinite data streams
+. Full-batch GD, on the other hand, would require multiple passes
+through a giant dataset per update (or a complex distributed memory
+system), which is often infeasible.
-
-- Steep coordinates require a smaller step size to avoid oscillation.
-- Flat/shallow coordinates could use a larger step to speed up progress.
-- This issue is pronounced in high-dimensional problems with **sparse or varying-scale features** – we need a method to adjust step sizesper feature.
-
+
+There is also a secondary memory effect: computing a full-batch
+gradient in deep learning requires storing all intermediate
+activations for backpropagation across the entire batch. A very large
+batch (approaching the full dataset) might exhaust GPU memory due to
+the need to hold activation gradients for thousands or millions of
+examples simultaneously. SGD/minibatches mitigate this by splitting
+the workload – e.g. with a mini-batch of size 32 or 256, memory use
+stays bounded, whereas a full-batch (size = \( N \)) forward/backward pass
+could not even be executed if \( N \) is huge. Techniques like gradient
+accumulation exist to simulate large-batch GD by summing many
+small-batch gradients – but these still process data in manageable
+chunks to avoid memory overflow. In summary, memory complexity for GD
+grows with \( N \), while for SGD it remains \( O(1) \) w.r.t. dataset size
+(only the model and perhaps a mini-batch reside in memory) . This is a
+key reason why batch GD “does not scale” to very large data and why
+virtually all large-scale machine learning algorithms rely on
+stochastic or mini-batch methods.
+
+
diff --git a/doc/pub/week37/html/._week37-bs035.html b/doc/pub/week37/html/._week37-bs035.html
index f133f3fde..bb8b33729 100644
--- a/doc/pub/week37/html/._week37-bs035.html
+++ b/doc/pub/week37/html/._week37-bs035.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,22 +374,78 @@ MathJax.Hub.Config({
-Motivation for Adaptive Step Sizes
+Empirical Evidence: Convergence Time and Memory in Practice
-
-- Instead of a fixed global \( \eta \), use an adaptive learning rate for each parameter that depends on the history of gradients.
-- Parameters that have large accumulated gradient magnitude should get smaller steps (they've been changing a lot), whereas parameters with small or infrequent gradients can have larger relative steps.
-- This is especially useful for sparse features: Rarely active features accumulate little gradient, so their learning rate remains comparatively high, ensuring they are not neglected
-- Conversely, frequently active features accumulate large gradient sums, and their learning rate automatically decreases, preventing too-large updates
-- Several algorithms implement this idea (AdaGrad, RMSProp, AdaDelta, Adam, etc.). We will derive **AdaGrad**, one of the first adaptive methods.
-
-
+Empirical studies strongly support the theoretical trade-offs
+above. In large-scale machine learning tasks, SGD often converges to a
+good solution much faster in wall-clock time than full-batch GD, and
+it uses far less memory. For example, Bottou & Bousquet (2008)
+analyzed learning time under a fixed computational budget and
+concluded that when data is abundant, it’s better to use a faster
+(even if less precise) optimization method to process more examples in
+the same time . This analysis showed that for large-scale problems,
+processing more data with SGD yields lower error than spending the
+time to do exact (batch) optimization on fewer data . In other words,
+if you have a time budget, it’s often optimal to accept slightly
+slower convergence per step (as with SGD) in exchange for being able
+to use many more training samples in that time. This phenomenon is
+borne out by experiments:
+
+Deep Neural Networks
-
-
-
-
-
+In modern deep learning, full-batch GD is so slow that it is rarely
+attempted; instead, mini-batch SGD is standard. A recent study
+demonstrated that it is possible to train a ResNet-50 on ImageNet
+using full-batch gradient descent, but it required careful tuning
+(e.g. gradient clipping, tiny learning rates) and vast computational
+resources – and even then, each full-batch update was extremely
+expensive.
+
+
+Using a huge batch
+(closer to full GD) tends to slow down convergence if the learning
+rate is not scaled up, and often encounters optimization difficulties
+(plateaus) that small batches avoid.
+Empirically, small or medium
+batch SGD finds minima in fewer clock hours because it can rapidly
+loop over the data with gradient noise aiding exploration.
+
+Memory constraints
+
+From a memory standpoint, practitioners note that batch GD becomes
+infeasible on large data. For example, if one tried to do full-batch
+training on a dataset that doesn’t fit in RAM or GPU memory, the
+program would resort to heavy disk I/O or simply crash. SGD
+circumvents this by processing mini-batches. Even in cases where data
+does fit in memory, using a full batch can spike memory usage due to
+storing all gradients. One empirical observation is that mini-batch
+training has a “lower, fluctuating usage pattern” of memory, whereas
+full-batch loading “quickly consumes memory (often exceeding limits)”
+. This is especially relevant for graph neural networks or other
+models where a “batch” may include a huge chunk of a graph: full-batch
+gradient computation can exhaust GPU memory, whereas mini-batch
+methods keep memory usage manageable .
+
+
+In summary, SGD converges faster than full-batch GD in terms of actual
+training time for large-scale problems, provided we measure
+convergence as reaching a good-enough solution. Theoretical bounds
+show SGD needs more iterations, but because it performs many more
+updates per unit time (and requires far less memory), it often
+achieves lower loss in a given time frame than GD. Full-batch GD might
+take slightly fewer iterations in theory, but each iteration is so
+costly that it is “slower… especially for large datasets” . Meanwhile,
+memory scaling strongly favors SGD: GD’s memory cost grows with
+dataset size, making it impractical beyond a point, whereas SGD’s
+memory use is modest and mostly constant w.r.t. \( N \) . These
+differences have made SGD (and mini-batch variants) the de facto
+choice for training large machine learning models, from logistic
+regression on millions of examples to deep neural networks with
+billions of parameters. The consensus in both research and practice is
+that for large-scale or high-dimensional tasks, SGD-type methods
+converge quicker per unit of computation and handle memory constraints
+better than standard full-batch gradient descent .
+
@@ -388,7 +472,7 @@ MathJax.Hub.Config({
44
45
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs036.html b/doc/pub/week37/html/._week37-bs036.html
index a6ecba5f5..c05a3e8cb 100644
--- a/doc/pub/week37/html/._week37-bs036.html
+++ b/doc/pub/week37/html/._week37-bs036.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,28 +374,30 @@ MathJax.Hub.Config({
-Derivation of the AdaGrad Algorithm
+Second moment of the gradient
-
-
-
-
-- AdaGrad maintains a running sum of squared gradients for each parameter (coordinate)
-- Let \( g_t = \nabla C_{i_t}(x_t) \) be the gradient at step \( t \) (or a subgradient for nondifferentiable cases).
-- Initialize \( r_0 = 0 \) (an all-zero vector in \( \mathbb{R}^d \)).
-- At each iteration \( t \), update the accumulation:
-
-$$
-r_t = r_{t-1} + g_t \circ g_t,
-$$
-
-
-- Here \( g_t \circ g_t \) denotes element-wise square of the gradient vector. \( g_t^{(j)} = g_{t-1}^{(j)} + (g_{t,j})^2 \) for each parameter \( j \).
-- We can view \( H_t = \mathrm{diag}(r_t) \) as a diagonal matrix of past squared gradients. Initially \( H_0 = 0 \).
-
-
-
+In stochastic gradient descent, with and without momentum, we still
+have to specify a schedule for tuning the learning rates \( \eta_t \)
+as a function of time. As discussed in the context of Newton's
+method, this presents a number of dilemmas. The learning rate is
+limited by the steepest direction which can change depending on the
+current position in the landscape. To circumvent this problem, ideally
+our algorithm would keep track of curvature and take large steps in
+shallow, flat directions and small steps in steep, narrow directions.
+Second-order methods accomplish this by calculating or approximating
+the Hessian and normalizing the learning rate by the
+curvature. However, this is very computationally expensive for
+extremely large models. Ideally, we would like to be able to
+adaptively change the step size to match the landscape without paying
+the steep computational price of calculating or approximating
+Hessians.
+
+During the last decade a number of methods have been introduced that accomplish
+this by tracking not only the gradient, but also the second moment of
+the gradient. These methods include AdaGrad, AdaDelta, Root Mean Squared Propagation (RMS-Prop), and
+ADAM.
+
@@ -394,7 +424,7 @@ $$
45
46
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs037.html b/doc/pub/week37/html/._week37-bs037.html
index 3a0721317..4988bb907 100644
--- a/doc/pub/week37/html/._week37-bs037.html
+++ b/doc/pub/week37/html/._week37-bs037.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,27 +374,20 @@ MathJax.Hub.Config({
-AdaGrad Update Rule Derivation
-
-We scale the gradient by the inverse square root of the accumulated matrix \( H_t \). The AdaGrad update at step \( t \) is:
-$$
-\theta_{t+1} =\theta_t - \eta H_t^{-1/2} g_t,
-$$
-
-where \( H_t^{-1/2} \) is the diagonal matrix with entries \( (r_{t}^{(1)})^{-1/2}, \dots, (r_{t}^{(d)})^{-1/2} \)
-In coordinates, this means each parameter \( j \) has an individual step size:
+
Challenge: Choosing a Fixed Learning Rate
+A fixed \( \eta \) is hard to get right:
+
+- If \( \eta \) is too large, the updates can overshoot the minimum, causing oscillations or divergence
+- If \( \eta \) is too small, convergence is very slow (many iterations to make progress)
+
+In practice, one often uses trial-and-error or schedules (decaying \( \eta \) over time) to find a workable balance.
+For a function with steep directions and flat directions, a single global \( \eta \) may be inappropriate:
-$$
- \theta_{t+1,j} =\theta_{t,j} -\frac{\eta}{\sqrt{r_{t,j}}}g_{t,j}.
-$$
-
-In practice we add a small constant \( \epsilon \) in the denominator for numerical stability to avoid division by zero:
-$$
-\theta_{t+1,j}= \theta_{t,j}-\frac{\eta}{\sqrt{\epsilon + r_{t,j}}}g_{t,j}.
-$$
-
-Equivalently, the effective learning rate for parameter \( j \) at time \( t \) is \( \displaystyle \alpha_{t,j} = \frac{\eta}{\sqrt{\epsilon + r_{t,j}}} \). This decreases over time as \( r_{t,j} \) grows.
-
+
+- Steep coordinates require a smaller step size to avoid oscillation.
+- Flat/shallow coordinates could use a larger step to speed up progress.
+- This issue is pronounced in high-dimensional problems with **sparse or varying-scale features** – we need a method to adjust step sizesper feature.
+
diff --git a/doc/pub/week37/html/._week37-bs038.html b/doc/pub/week37/html/._week37-bs038.html
index 3ef9f1765..d07855f29 100644
--- a/doc/pub/week37/html/._week37-bs038.html
+++ b/doc/pub/week37/html/._week37-bs038.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,18 +374,23 @@ MathJax.Hub.Config({
-AdaGrad Properties
+Motivation for Adaptive Step Sizes
-- AdaGrad automatically tunes the step size for each parameter. Parameters with more volatile or large gradients get smaller steps, and those with small or infrequent gradients get relatively larger steps
-- No manual schedule needed: The accumulation \( r_t \) keeps increasing (or stays the same if gradient is zero), so step sizes \( \eta/\sqrt{r_t} \) are non-increasing. This has a similar effect to a learning rate schedule, but individualized per coordinate.
-- Sparse data benefit: For very sparse features, \( r_{t,j} \) grows slowly, so that feature’s parameter retains a higher learning rate for longer, allowing it to make significant updates when it does get a gradient signal
-- Convergence: In convex optimization, AdaGrad can be shown to achieve a sub-linear convergence rate comparable to the best fixed learning rate tuned for the problem
-
-It effectively reduces the need to tune \( \eta \) by hand.
-
-- Limitations: Because \( r_t \) accumulates without bound, AdaGrad’s learning rates can become extremely small over long training, potentially slowing progress. (Later variants like RMSProp, AdaDelta, Adam address this by modifying the accumulation rule.)
+- Instead of a fixed global \( \eta \), use an adaptive learning rate for each parameter that depends on the history of gradients.
+- Parameters that have large accumulated gradient magnitude should get smaller steps (they've been changing a lot), whereas parameters with small or infrequent gradients can have larger relative steps.
+- This is especially useful for sparse features: Rarely active features accumulate little gradient, so their learning rate remains comparatively high, ensuring they are not neglected
+- Conversely, frequently active features accumulate large gradient sums, and their learning rate automatically decreases, preventing too-large updates
+- Several algorithms implement this idea (AdaGrad, RMSProp, AdaDelta, Adam, etc.). We will derive **AdaGrad**, one of the first adaptive methods.
+
+
+
+
+
+
+
+
diff --git a/doc/pub/week37/html/._week37-bs039.html b/doc/pub/week37/html/._week37-bs039.html
index 5a2a44aa1..fb28b394d 100644
--- a/doc/pub/week37/html/._week37-bs039.html
+++ b/doc/pub/week37/html/._week37-bs039.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,29 +374,28 @@ MathJax.Hub.Config({
-RMSProp: Adaptive Learning Rates
+Derivation of the AdaGrad Algorithm
-Addresses AdaGrad’s diminishing learning rate issue.
-Uses a decaying average of squared gradients (instead of a cumulative sum):
-
-$$
-v_t = \rho v_{t-1} + (1-\rho)(\nabla C(\theta_t))^2,
-$$
-
-with \( \rho \) typically \( 0.9 \) (or \( 0.99 \)).
+
+
+
-- Update: \( \theta_{t+1} = \theta_t - \frac{\eta}{\sqrt{v_t + \epsilon}} \nabla C(\theta_t) \).
-- Recent gradients have more weight, so \( v_t \) adapts to the current landscape.
-- Avoids AdaGrad’s “infinite memory” problem – learning rate does not continuously decay to zero.
+- AdaGrad maintains a running sum of squared gradients for each parameter (coordinate)
+- Let \( g_t = \nabla C_{i_t}(x_t) \) be the gradient at step \( t \) (or a subgradient for nondifferentiable cases).
+- Initialize \( r_0 = 0 \) (an all-zero vector in \( \mathbb{R}^d \)).
+- At each iteration \( t \), update the accumulation:
-
RMSProp was first proposed in lecture notes by Geoff Hinton, 2012 – unpublished.)
-
+$$
+r_t = r_{t-1} + g_t \circ g_t,
+$$
+
+
+- Here \( g_t \circ g_t \) denotes element-wise square of the gradient vector. \( g_t^{(j)} = g_{t-1}^{(j)} + (g_{t,j})^2 \) for each parameter \( j \).
+- We can view \( H_t = \mathrm{diag}(r_t) \) as a diagonal matrix of past squared gradients. Initially \( H_0 = 0 \).
+
+
+
-
-
-
-
-
@@ -395,7 +422,7 @@ $$
48
49
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs040.html b/doc/pub/week37/html/._week37-bs040.html
index de7c10a3c..2ffa9b8af 100644
--- a/doc/pub/week37/html/._week37-bs040.html
+++ b/doc/pub/week37/html/._week37-bs040.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,17 +374,26 @@ MathJax.Hub.Config({
-Adam Optimizer
+AdaGrad Update Rule Derivation
-Why combine Momentum and RMSProp? Motivation for Adam: Adaptive Moment Estimation (Adam) was introduced by Kingma an Ba (2014) to combine the benefits of momentum and RMSProp.
+We scale the gradient by the inverse square root of the accumulated matrix \( H_t \). The AdaGrad update at step \( t \) is:
+$$
+\theta_{t+1} =\theta_t - \eta H_t^{-1/2} g_t,
+$$
-
-- Fast convergence by smoothing gradients (accelerates in long-term gradient direction).
-- Adaptive rates (RMSProp): Per-dimension learning rate scaling for stability (handles different feature scales, sparse gradients).
-- Adam uses both: maintains moving averages of both first moment (gradients) and second moment (squared gradients)
-- Additionally, includes a mechanism to correct the bias in these moving averages (crucial in early iterations)
-
-Result: Adam is robust, achieves faster convergence with less tuning, and often outperforms SGD (with momentum) in practice.
+where \( H_t^{-1/2} \) is the diagonal matrix with entries \( (r_{t}^{(1)})^{-1/2}, \dots, (r_{t}^{(d)})^{-1/2} \)
+In coordinates, this means each parameter \( j \) has an individual step size:
+
+$$
+ \theta_{t+1,j} =\theta_{t,j} -\frac{\eta}{\sqrt{r_{t,j}}}g_{t,j}.
+$$
+
+In practice we add a small constant \( \epsilon \) in the denominator for numerical stability to avoid division by zero:
+$$
+\theta_{t+1,j}= \theta_{t,j}-\frac{\eta}{\sqrt{\epsilon + r_{t,j}}}g_{t,j}.
+$$
+
+Equivalently, the effective learning rate for parameter \( j \) at time \( t \) is \( \displaystyle \alpha_{t,j} = \frac{\eta}{\sqrt{\epsilon + r_{t,j}}} \). This decreases over time as \( r_{t,j} \) grows.
@@ -383,7 +420,7 @@ MathJax.Hub.Config({
49
50
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs041.html b/doc/pub/week37/html/._week37-bs041.html
index 4eca5ec84..acba4fc7d 100644
--- a/doc/pub/week37/html/._week37-bs041.html
+++ b/doc/pub/week37/html/._week37-bs041.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,17 +374,18 @@ MathJax.Hub.Config({
-
-
-In ADAM, we keep a running average of
-both the first and second moment of the gradient and use this
-information to adaptively change the learning rate for different
-parameters. The method is efficient when working with large
-problems involving lots data and/or parameters. It is a combination of the
-gradient descent with momentum algorithm and the RMSprop algorithm
-discussed above.
-
+AdaGrad Properties
+
+- AdaGrad automatically tunes the step size for each parameter. Parameters with more volatile or large gradients get smaller steps, and those with small or infrequent gradients get relatively larger steps
+- No manual schedule needed: The accumulation \( r_t \) keeps increasing (or stays the same if gradient is zero), so step sizes \( \eta/\sqrt{r_t} \) are non-increasing. This has a similar effect to a learning rate schedule, but individualized per coordinate.
+- Sparse data benefit: For very sparse features, \( r_{t,j} \) grows slowly, so that feature’s parameter retains a higher learning rate for longer, allowing it to make significant updates when it does get a gradient signal
+- Convergence: In convex optimization, AdaGrad can be shown to achieve a sub-linear convergence rate comparable to the best fixed learning rate tuned for the problem
+
+It effectively reduces the need to tune \( \eta \) by hand.
+
+- Limitations: Because \( r_t \) accumulates without bound, AdaGrad’s learning rates can become extremely small over long training, potentially slowing progress. (Later variants like RMSProp, AdaDelta, Adam address this by modifying the accumulation rule.)
+
diff --git a/doc/pub/week37/html/._week37-bs042.html b/doc/pub/week37/html/._week37-bs042.html
index 074d9ee9b..e580daf68 100644
--- a/doc/pub/week37/html/._week37-bs042.html
+++ b/doc/pub/week37/html/._week37-bs042.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,15 +374,29 @@ MathJax.Hub.Config({
-Why Combine Momentum and RMSProp?
+RMSProp: Adaptive Learning Rates
+Addresses AdaGrad’s diminishing learning rate issue.
+Uses a decaying average of squared gradients (instead of a cumulative sum):
+
+$$
+v_t = \rho v_{t-1} + (1-\rho)(\nabla C(\theta_t))^2,
+$$
+
+with \( \rho \) typically \( 0.9 \) (or \( 0.99 \)).
-- Momentum: Fast convergence by smoothing gradients (accelerates in long-term gradient direction).
-- Adaptive rates (RMSProp): Per-dimension learning rate scaling for stability (handles different feature scales, sparse gradients).
-- Adam uses both: maintains moving averages of both first moment (gradients) and second moment (squared gradients)
-- Additionally, includes a mechanism to correct the bias in these moving averages (crucial in early iterations)
+- Update: \( \theta_{t+1} = \theta_t - \frac{\eta}{\sqrt{v_t + \epsilon}} \nabla C(\theta_t) \).
+- Recent gradients have more weight, so \( v_t \) adapts to the current landscape.
+- Avoids AdaGrad’s “infinite memory” problem – learning rate does not continuously decay to zero.
-Result: Adam is robust, achieves faster convergence with less tuning, and often outperforms SGD (with momentum) in practice
+RMSProp was first proposed in lecture notes by Geoff Hinton, 2012 – unpublished.)
+
+
+
+
+
+
+
@@ -381,7 +423,7 @@ MathJax.Hub.Config({
51
52
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs043.html b/doc/pub/week37/html/._week37-bs043.html
index 799815721..558b02294 100644
--- a/doc/pub/week37/html/._week37-bs043.html
+++ b/doc/pub/week37/html/._week37-bs043.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,31 +374,17 @@ MathJax.Hub.Config({
-Adam: Exponential Moving Averages (Moments)
-Adam maintains two moving averages at each time step \( t \) for each parameter \( w \):
-
-
-
-
The Momentum term
-$$
-m_t = \beta_1m_{t-1} + (1-\beta_1)\, \nabla C(\theta_t),
-$$
-
-
+Adam Optimizer
-
-
-
-
The RMS term
-$$
-v_t = \beta_2v_{t-1} + (1-\beta_2)(\nabla C(\theta_t))^2,
-$$
+
Why combine Momentum and RMSProp? Motivation for Adam: Adaptive Moment Estimation (Adam) was introduced by Kingma an Ba (2014) to combine the benefits of momentum and RMSProp.
-
with typical \( \beta_1 = 0.9 \), \( \beta_2 = 0.999 \). Initialize \( m_0 = 0 \), \( v_0 = 0 \).
-
-
-
- These are biased estimators of the true first and second moment of the gradients, especially at the start (since \( m_0,v_0 \) are zero)
+
+- Fast convergence by smoothing gradients (accelerates in long-term gradient direction).
+- Adaptive rates (RMSProp): Per-dimension learning rate scaling for stability (handles different feature scales, sparse gradients).
+- Adam uses both: maintains moving averages of both first moment (gradients) and second moment (squared gradients)
+- Additionally, includes a mechanism to correct the bias in these moving averages (crucial in early iterations)
+
+Result: Adam is robust, achieves faster convergence with less tuning, and often outperforms SGD (with momentum) in practice.
@@ -397,7 +411,7 @@ $$
52
53
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs044.html b/doc/pub/week37/html/._week37-bs044.html
index 3f15d89ab..7969230b4 100644
--- a/doc/pub/week37/html/._week37-bs044.html
+++ b/doc/pub/week37/html/._week37-bs044.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,17 +374,17 @@ MathJax.Hub.Config({
-Adam: Bias Correction
-To counteract initialization bias in \( m_t, v_t \), Adam computes bias-corrected estimates
-$$
-\hat{m}_t = \frac{m_t}{1 - \beta_1^t}, \qquad \hat{v}_t = \frac{v_t}{1 - \beta_2^t}.
-$$
+
+
+In ADAM, we keep a running average of
+both the first and second moment of the gradient and use this
+information to adaptively change the learning rate for different
+parameters. The method is efficient when working with large
+problems involving lots data and/or parameters. It is a combination of the
+gradient descent with momentum algorithm and the RMSprop algorithm
+discussed above.
+
-
-- When \( t \) is small, \( 1-\beta_i^t \approx 0 \), so \( \hat{m}_t, \hat{v}_t \) significantly larger than raw \( m_t, v_t \), compensating for the initial zero bias.
-- As \( t \) increases, \( 1-\beta_i^t \to 1 \), and \( \hat{m}_t, \hat{v}_t \) converge to \( m_t, v_t \).
-- Bias correction is important for Adam’s stability in early iterations
-
diff --git a/doc/pub/week37/html/._week37-bs045.html b/doc/pub/week37/html/._week37-bs045.html
index 530ca9893..66740a39a 100644
--- a/doc/pub/week37/html/._week37-bs045.html
+++ b/doc/pub/week37/html/._week37-bs045.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,23 +374,15 @@ MathJax.Hub.Config({
-Adam: Update Rule Derivation
-Finally, Adam updates parameters using the bias-corrected moments:
-$$
-\theta_{t+1} =\theta_t -\frac{\alpha}{\sqrt{\hat{v}_t} + \epsilon}\hat{m}_t,
-$$
+Why Combine Momentum and RMSProp?
-where \( \epsilon \) is a small constant (e.g. \( 10^{-8} \)) to prevent division by zero.
-Breaking it down:
-
-- Compute gradient \( \nabla C(\theta_t) \).
-- Update first moment \( m_t \) and second moment \( v_t \) (exponential moving averages).
-- Bias-correct: \( \hat{m}_t = m_t/(1-\beta_1^t) \), \( \; \hat{v}_t = v_t/(1-\beta_2^t) \).
-- Compute step: \( \Delta \theta_t = \frac{\hat{m}_t}{\sqrt{\hat{v}_t} + \epsilon} \).
-- Update parameters: \( \theta_{t+1} = \theta_t - \alpha\, \Delta \theta_t \).
+- Momentum: Fast convergence by smoothing gradients (accelerates in long-term gradient direction).
+- Adaptive rates (RMSProp): Per-dimension learning rate scaling for stability (handles different feature scales, sparse gradients).
+- Adam uses both: maintains moving averages of both first moment (gradients) and second moment (squared gradients)
+- Additionally, includes a mechanism to correct the bias in these moving averages (crucial in early iterations)
-This is the Adam update rule as given in the original paper.
+Result: Adam is robust, achieves faster convergence with less tuning, and often outperforms SGD (with momentum) in practice
@@ -389,7 +409,7 @@ Breaking it down:
54
55
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs046.html b/doc/pub/week37/html/._week37-bs046.html
index 4618a100c..50fa05f9a 100644
--- a/doc/pub/week37/html/._week37-bs046.html
+++ b/doc/pub/week37/html/._week37-bs046.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,19 +374,31 @@ MathJax.Hub.Config({
-Adam vs. AdaGrad and RMSProp
+Adam: Exponential Moving Averages (Moments)
+Adam maintains two moving averages at each time step \( t \) for each parameter \( w \):
+
+
+
+
The Momentum term
+$$
+m_t = \beta_1m_{t-1} + (1-\beta_1)\, \nabla C(\theta_t),
+$$
+
+
-
-- AdaGrad: Uses per-coordinate scaling like Adam, but no momentum. Tends to slow down too much due to cumulative history (no forgetting)
-- RMSProp: Uses moving average of squared gradients (like Adam’s \( v_t \)) to maintain adaptive learning rates, but does not include momentum or bias-correction.
-- Adam: Effectively RMSProp + Momentum + Bias-correction
-
- - Momentum (\( m_t \)) provides acceleration and smoother convergence.
- - Adaptive \( v_t \) scaling moderates the step size per dimension.
- - Bias correction (absent in AdaGrad/RMSProp) ensures robust estimates early on.
-
-
-In practice, Adam often yields faster convergence and better tuning stability than RMSProp or AdaGrad alone
+
+
+
+
The RMS term
+$$
+v_t = \beta_2v_{t-1} + (1-\beta_2)(\nabla C(\theta_t))^2,
+$$
+
+
with typical \( \beta_1 = 0.9 \), \( \beta_2 = 0.999 \). Initialize \( m_0 = 0 \), \( v_0 = 0 \).
+
+
+
+ These are biased estimators of the true first and second moment of the gradients, especially at the start (since \( m_0,v_0 \) are zero)
@@ -385,7 +425,7 @@ MathJax.Hub.Config({
55
56
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs047.html b/doc/pub/week37/html/._week37-bs047.html
index 76b5accf8..bfd6fc51b 100644
--- a/doc/pub/week37/html/._week37-bs047.html
+++ b/doc/pub/week37/html/._week37-bs047.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,21 +374,17 @@ MathJax.Hub.Config({
-Adaptivity Across Dimensions
-
-
-- Adam adapts the step size \emph{per coordinate}: parameters with larger gradient variance get smaller effective steps, those with smaller or sparse gradients get larger steps.
-- This per-dimension adaptivity is inherited from AdaGrad/RMSProp and helps handle ill-conditioned or sparse problems.
-- Meanwhile, momentum (first moment) allows Adam to continue making progress even if gradients become small or noisy, by leveraging accumulated direction.
-
-
-
-
-
-
-
-
+Adam: Bias Correction
+To counteract initialization bias in \( m_t, v_t \), Adam computes bias-corrected estimates
+$$
+\hat{m}_t = \frac{m_t}{1 - \beta_1^t}, \qquad \hat{v}_t = \frac{v_t}{1 - \beta_2^t}.
+$$
+
+- When \( t \) is small, \( 1-\beta_i^t \approx 0 \), so \( \hat{m}_t, \hat{v}_t \) significantly larger than raw \( m_t, v_t \), compensating for the initial zero bias.
+- As \( t \) increases, \( 1-\beta_i^t \to 1 \), and \( \hat{m}_t, \hat{v}_t \) converge to \( m_t, v_t \).
+- Bias correction is important for Adam’s stability in early iterations
+
diff --git a/doc/pub/week37/html/._week37-bs048.html b/doc/pub/week37/html/._week37-bs048.html
index d1a42f7ed..0cf9bdb3c 100644
--- a/doc/pub/week37/html/._week37-bs048.html
+++ b/doc/pub/week37/html/._week37-bs048.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,11 +374,23 @@ MathJax.Hub.Config({
-Algorithms and codes for Adagrad, RMSprop and Adam
+Adam: Update Rule Derivation
+Finally, Adam updates parameters using the bias-corrected moments:
+$$
+\theta_{t+1} =\theta_t -\frac{\alpha}{\sqrt{\hat{v}_t} + \epsilon}\hat{m}_t,
+$$
-The algorithms we have implemented are well described in the text by Goodfellow, Bengio and Courville, chapter 8.
-
-The codes which implement these algorithms are discussed below here.
+where \( \epsilon \) is a small constant (e.g. \( 10^{-8} \)) to prevent division by zero.
+Breaking it down:
+
+
+- Compute gradient \( \nabla C(\theta_t) \).
+- Update first moment \( m_t \) and second moment \( v_t \) (exponential moving averages).
+- Bias-correct: \( \hat{m}_t = m_t/(1-\beta_1^t) \), \( \; \hat{v}_t = v_t/(1-\beta_2^t) \).
+- Compute step: \( \Delta \theta_t = \frac{\hat{m}_t}{\sqrt{\hat{v}_t} + \epsilon} \).
+- Update parameters: \( \theta_{t+1} = \theta_t - \alpha\, \Delta \theta_t \).
+
+This is the Adam update rule as given in the original paper.
@@ -377,7 +417,7 @@ MathJax.Hub.Config({
57
58
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/._week37-bs049.html b/doc/pub/week37/html/._week37-bs049.html
index cb5236451..4eb424b04 100644
--- a/doc/pub/week37/html/._week37-bs049.html
+++ b/doc/pub/week37/html/._week37-bs049.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -346,14 +374,20 @@ MathJax.Hub.Config({
-Practical tips
+Adam vs. AdaGrad and RMSProp
+
+- AdaGrad: Uses per-coordinate scaling like Adam, but no momentum. Tends to slow down too much due to cumulative history (no forgetting)
+- RMSProp: Uses moving average of squared gradients (like Adam’s \( v_t \)) to maintain adaptive learning rates, but does not include momentum or bias-correction.
+- Adam: Effectively RMSProp + Momentum + Bias-correction
-- Randomize the data when making mini-batches. It is always important to randomly shuffle the data when forming mini-batches. Otherwise, the gradient descent method can fit spurious correlations resulting from the order in which data is presented.
-- Transform your inputs. Learning becomes difficult when our landscape has a mixture of steep and flat directions. One simple trick for minimizing these situations is to standardize the data by subtracting the mean and normalizing the variance of input variables. Whenever possible, also decorrelate the inputs. To understand why this is helpful, consider the case of linear regression. It is easy to show that for the squared error cost function, the Hessian of the cost function is just the correlation matrix between the inputs. Thus, by standardizing the inputs, we are ensuring that the landscape looks homogeneous in all directions in parameter space. Since most deep networks can be viewed as linear transformations followed by a non-linearity at each layer, we expect this intuition to hold beyond the linear case.
-- Monitor the out-of-sample performance. Always monitor the performance of your model on a validation set (a small portion of the training data that is held out of the training process to serve as a proxy for the test set. If the validation error starts increasing, then the model is beginning to overfit. Terminate the learning process. This early stopping significantly improves performance in many settings.
-- Adaptive optimization methods don't always have good generalization. Recent studies have shown that adaptive methods such as ADAM, RMSPorp, and AdaGrad tend to have poor generalization compared to SGD or SGD with momentum, particularly in the high-dimensional limit (i.e. the number of parameters exceeds the number of data points). Although it is not clear at this stage why these methods perform so well in training deep neural networks, simpler procedures like properly-tuned SGD may work as well or better in these applications.
+ - Momentum (\( m_t \)) provides acceleration and smoother convergence.
+ - Adaptive \( v_t \) scaling moderates the step size per dimension.
+ - Bias correction (absent in AdaGrad/RMSProp) ensures robust estimates early on.
+
+In practice, Adam often yields faster convergence and better tuning stability than RMSProp or AdaGrad alone
+
diff --git a/doc/pub/week37/html/week37-bs.html b/doc/pub/week37/html/week37-bs.html
index 6ff6b267b..5cb7bff40 100644
--- a/doc/pub/week37/html/week37-bs.html
+++ b/doc/pub/week37/html/week37-bs.html
@@ -114,6 +114,26 @@ doconce format html week37.do.txt --html_style=bootstrap --pygments_html_style=d
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -270,71 +290,79 @@ MathJax.Hub.Config({
Contents
@@ -388,7 +416,7 @@ MathJax.Hub.Config({
9
10
...
- 63
+ 66
»
diff --git a/doc/pub/week37/html/week37-reveal.html b/doc/pub/week37/html/week37-reveal.html
index 9d2fd01d4..239b77715 100644
--- a/doc/pub/week37/html/week37-reveal.html
+++ b/doc/pub/week37/html/week37-reveal.html
@@ -1181,6 +1181,253 @@ useful.
+
+SGD vs Full-Batch GD: Convergence Speed and Memory Comparison
+Theoretical Convergence Speed and convex optimization
+
+Consider minimizing an empirical cost function
+
+$$
+C(\theta) =\frac{1}{N}\sum_{i=1}^N l_i(\theta),
+$$
+
+
+
where each \( l_i(\theta) \) is a
+differentiable loss term. Gradient Descent (GD) updates parameters
+using the full gradient \( \nabla C(\theta) \), while Stochastic Gradient
+Descent (SGD) uses a single sample (or mini-batch) gradient \( \nabla
+l_i(\theta) \) selected at random. In equation form, one GD step is:
+
+
+
+$$
+\theta_{t+1} = \theta_t-\eta \nabla C(\theta_t) =\theta_t -\eta \frac{1}{N}\sum_{i=1}^N \nabla l_i(\theta_t),
+$$
+
+
+
whereas one SGD step is:
+
+
+$$
+\theta_{t+1} = \theta_t -\eta \nabla l_{i_t}(\theta_t),
+$$
+
+
+
with \( i_t \) randomly chosen. On smooth convex problems, GD and SGD both
+converge to the global minimum, but their rates differ. GD can take
+larger, more stable steps since it uses the exact gradient, achieving
+an error that decreases on the order of \( O(1/t) \) per iteration for
+convex objectives (and even exponentially fast for strongly convex
+cases). In contrast, plain SGD has more variance in each step, leading
+to sublinear convergence in expectation – typically \( O(1/\sqrt{t}) \)
+for general convex objectives (\thetaith appropriate diminishing step
+sizes) . Intuitively, GD’s trajectory is smoother and more
+predictable, while SGD’s path oscillates due to noise but costs far
+less per iteration, enabling many more updates in the same time.
+
+Strongly Convex Case
+
+If \( C(\theta) \) is strongly convex and \( L \)-smooth (so GD enjoys linear
+convergence), the gap \( C(\theta_t)-C(\theta^*) \) for GD shrinks as
+
+
+$$
+C(\theta_t) - C(\theta^* ) \le \Big(1 - \frac{\mu}{L}\Big)^t [C(\theta_0)-C(\theta^*)],
+$$
+
+
+
a geometric (linear) convergence per iteration . Achieving an
+\( \epsilon \)-accurate solution thus takes on the order of
+\( \log(1/\epsilon) \) iterations for GD. However, each GD iteration costs
+\( O(N) \) gradient evaluations. SGD cannot exploit strong convexity to
+obtain a linear rate – instead, with a properly decaying step size
+(e.g. \( \eta_t = \frac{1}{\mu t} \)) or iterate averaging, SGD attains an
+\( O(1/t) \) convergence rate in expectation . For example, one result
+of Moulines and Bach 2011, see https://papers.nips.cc/paper_files/paper/2011/hash/40008b9a5380fcacce3976bf7c08af5b-Abstract.html shows that with \( \eta_t = \Theta(1/t) \),
+
+
+$$
+\mathbb{E}[C(\theta_t) - C(\theta^*)] = O(1/t),
+$$
+
+
+
for strongly convex, smooth \( F \) . This \( 1/t \) rate is slower per
+iteration than GD’s exponential decay, but each SGD iteration is \( N \)
+times cheaper. In fact, to reach error \( \epsilon \), plain SGD needs on
+the order of \( T=O(1/\epsilon) \) iterations (sub-linear convergence),
+while GD needs \( O(\log(1/\epsilon)) \) iterations. When accounting for
+cost-per-iteration, GD requires \( O(N \log(1/\epsilon)) \) total gradient
+computations versus SGD’s \( O(1/\epsilon) \) single-sample
+computations. In large-scale regimes (huge \( N \)), SGD can be
+faster in wall-clock time because \( N \log(1/\epsilon) \) may far exceed
+\( 1/\epsilon \) for reasonable accuracy levels. In other words,
+with millions of data points, one epoch of GD (one full gradient) is
+extremely costly, whereas SGD can make \( N \) cheap updates in the time
+GD makes one – often yielding a good solution faster in practice, even
+though SGD’s asymptotic error decays more slowly. As one lecture
+succinctly puts it: “SGD can be super effective in terms of iteration
+cost and memory, but SGD is slow to converge and can’t adapt to strong
+convexity” . Thus, the break-even point depends on \( N \) and the desired
+accuracy: for moderate accuracy on very large \( N \), SGD’s cheaper
+updates win; for extremely high precision (very small \( \epsilon \)) on a
+modest \( N \), GD’s fast convergence per step can be advantageous.
+
+Non-Convex Problems
+
+In non-convex optimization (e.g. deep neural networks), neither GD nor
+SGD guarantees global minima, but SGD often displays faster progress
+in finding useful minima. Theoretical results here are weaker, usually
+showing convergence to a stationary point \( \theta \) (\( |\nabla C| \) is
+small) in expectation. For example, GD might require \( O(1/\epsilon^2) \)
+iterations to ensure \( |\nabla C(\theta)| < \epsilon \), and SGD typically has
+similar polynomial complexity (often worse due to gradient
+noise). However, a noteworthy difference is that SGD’s stochasticity
+can help escape saddle points or poor local minima. Random gradient
+fluctuations act like implicit noise, helping the iterate “jump” out
+of flat saddle regions where full-batch GD could stagnate . In fact,
+research has shown that adding noise to GD can guarantee escaping
+saddle points in polynomial time, and the inherent noise in SGD often
+serves this role. Empirically, this means SGD can sometimes find a
+lower loss basin faster, whereas full-batch GD might get “stuck” near
+saddle points or need a very small learning rate to navigate complex
+error surfaces . Overall, in modern high-dimensional machine learning,
+SGD (or mini-batch SGD) is the workhorse for large non-convex problems
+because it converges to good solutions much faster in practice,
+despite the lack of a linear convergence guarantee. Full-batch GD is
+rarely used on large neural networks, as it would require tiny steps
+to avoid divergence and is extremely slow per iteration .
+
+
+
+
+Memory Usage and Scalability
+
+A major advantage of SGD is its memory efficiency in handling large
+datasets. Full-batch GD requires access to the entire training set for
+each iteration, which often means the whole dataset (or a large
+subset) must reside in memory to compute \( \nabla C(\theta) \) . This results
+in memory usage that scales linearly with the dataset size \( N \). For
+instance, if each training sample is large (e.g. high-dimensional
+features), computing a full gradient may require storing a substantial
+portion of the data or all intermediate gradients until they are
+aggregated. In contrast, SGD needs only a single (or a small
+mini-batch of) training example(s) in memory at any time . The
+algorithm processes one sample (or mini-batch) at a time and
+immediately updates the model, discarding that sample before moving to
+the next. This streaming approach means that memory footprint is
+essentially independent of \( N \) (apart from storing the model
+parameters themselves). As one source notes, gradient descent
+“requires more memory than SGD” because it “must store the entire
+dataset for each iteration,” whereas SGD “only needs to store the
+current training example” . In practical terms, if you have a dataset
+of size, say, 1 million examples, full-batch GD would need memory for
+all million every step, while SGD could be implemented to load just
+one example at a time – a crucial benefit if data are too large to fit
+in RAM or GPU memory. This scalability makes SGD suitable for
+large-scale learning: as long as you can stream data from disk, SGD
+can handle arbitrarily large datasets with fixed memory. In fact, SGD
+“does not need to remember which examples were visited” in the past,
+allowing it to run in an online fashion on infinite data streams
+. Full-batch GD, on the other hand, would require multiple passes
+through a giant dataset per update (or a complex distributed memory
+system), which is often infeasible.
+
+
+There is also a secondary memory effect: computing a full-batch
+gradient in deep learning requires storing all intermediate
+activations for backpropagation across the entire batch. A very large
+batch (approaching the full dataset) might exhaust GPU memory due to
+the need to hold activation gradients for thousands or millions of
+examples simultaneously. SGD/minibatches mitigate this by splitting
+the workload – e.g. with a mini-batch of size 32 or 256, memory use
+stays bounded, whereas a full-batch (size = \( N \)) forward/backward pass
+could not even be executed if \( N \) is huge. Techniques like gradient
+accumulation exist to simulate large-batch GD by summing many
+small-batch gradients – but these still process data in manageable
+chunks to avoid memory overflow. In summary, memory complexity for GD
+grows with \( N \), while for SGD it remains \( O(1) \) w.r.t. dataset size
+(only the model and perhaps a mini-batch reside in memory) . This is a
+key reason why batch GD “does not scale” to very large data and why
+virtually all large-scale machine learning algorithms rely on
+stochastic or mini-batch methods.
+
+
+
+
+Empirical Evidence: Convergence Time and Memory in Practice
+
+Empirical studies strongly support the theoretical trade-offs
+above. In large-scale machine learning tasks, SGD often converges to a
+good solution much faster in wall-clock time than full-batch GD, and
+it uses far less memory. For example, Bottou & Bousquet (2008)
+analyzed learning time under a fixed computational budget and
+concluded that when data is abundant, it’s better to use a faster
+(even if less precise) optimization method to process more examples in
+the same time . This analysis showed that for large-scale problems,
+processing more data with SGD yields lower error than spending the
+time to do exact (batch) optimization on fewer data . In other words,
+if you have a time budget, it’s often optimal to accept slightly
+slower convergence per step (as with SGD) in exchange for being able
+to use many more training samples in that time. This phenomenon is
+borne out by experiments:
+
+Deep Neural Networks
+
+In modern deep learning, full-batch GD is so slow that it is rarely
+attempted; instead, mini-batch SGD is standard. A recent study
+demonstrated that it is possible to train a ResNet-50 on ImageNet
+using full-batch gradient descent, but it required careful tuning
+(e.g. gradient clipping, tiny learning rates) and vast computational
+resources – and even then, each full-batch update was extremely
+expensive.
+
+
+Using a huge batch
+(closer to full GD) tends to slow down convergence if the learning
+rate is not scaled up, and often encounters optimization difficulties
+(plateaus) that small batches avoid.
+Empirically, small or medium
+batch SGD finds minima in fewer clock hours because it can rapidly
+loop over the data with gradient noise aiding exploration.
+
+Memory constraints
+
+From a memory standpoint, practitioners note that batch GD becomes
+infeasible on large data. For example, if one tried to do full-batch
+training on a dataset that doesn’t fit in RAM or GPU memory, the
+program would resort to heavy disk I/O or simply crash. SGD
+circumvents this by processing mini-batches. Even in cases where data
+does fit in memory, using a full batch can spike memory usage due to
+storing all gradients. One empirical observation is that mini-batch
+training has a “lower, fluctuating usage pattern” of memory, whereas
+full-batch loading “quickly consumes memory (often exceeding limits)”
+. This is especially relevant for graph neural networks or other
+models where a “batch” may include a huge chunk of a graph: full-batch
+gradient computation can exhaust GPU memory, whereas mini-batch
+methods keep memory usage manageable .
+
+
+In summary, SGD converges faster than full-batch GD in terms of actual
+training time for large-scale problems, provided we measure
+convergence as reaching a good-enough solution. Theoretical bounds
+show SGD needs more iterations, but because it performs many more
+updates per unit time (and requires far less memory), it often
+achieves lower loss in a given time frame than GD. Full-batch GD might
+take slightly fewer iterations in theory, but each iteration is so
+costly that it is “slower… especially for large datasets” . Meanwhile,
+memory scaling strongly favors SGD: GD’s memory cost grows with
+dataset size, making it impractical beyond a point, whereas SGD’s
+memory use is modest and mostly constant w.r.t. \( N \) . These
+differences have made SGD (and mini-batch variants) the de facto
+choice for training large machine learning models, from logistic
+regression on millions of examples to deep neural networks with
+billions of parameters. The consensus in both research and practice is
+that for large-scale or high-dimensional tasks, SGD-type methods
+converge quicker per unit of computation and handle memory constraints
+better than standard full-batch gradient descent .
+
+
+
Second moment of the gradient
diff --git a/doc/pub/week37/html/week37-solarized.html b/doc/pub/week37/html/week37-solarized.html
index d5d752058..1e222348d 100644
--- a/doc/pub/week37/html/week37-solarized.html
+++ b/doc/pub/week37/html/week37-solarized.html
@@ -141,6 +141,26 @@ div.toc p,a {
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -1233,6 +1253,240 @@ mini-batches. The discussion
useful.
+
+SGD vs Full-Batch GD: Convergence Speed and Memory Comparison
+Theoretical Convergence Speed and convex optimization
+
+Consider minimizing an empirical cost function
+$$
+C(\theta) =\frac{1}{N}\sum_{i=1}^N l_i(\theta),
+$$
+
+where each \( l_i(\theta) \) is a
+differentiable loss term. Gradient Descent (GD) updates parameters
+using the full gradient \( \nabla C(\theta) \), while Stochastic Gradient
+Descent (SGD) uses a single sample (or mini-batch) gradient \( \nabla
+l_i(\theta) \) selected at random. In equation form, one GD step is:
+
+
+$$
+\theta_{t+1} = \theta_t-\eta \nabla C(\theta_t) =\theta_t -\eta \frac{1}{N}\sum_{i=1}^N \nabla l_i(\theta_t),
+$$
+
+whereas one SGD step is:
+
+$$
+\theta_{t+1} = \theta_t -\eta \nabla l_{i_t}(\theta_t),
+$$
+
+with \( i_t \) randomly chosen. On smooth convex problems, GD and SGD both
+converge to the global minimum, but their rates differ. GD can take
+larger, more stable steps since it uses the exact gradient, achieving
+an error that decreases on the order of \( O(1/t) \) per iteration for
+convex objectives (and even exponentially fast for strongly convex
+cases). In contrast, plain SGD has more variance in each step, leading
+to sublinear convergence in expectation – typically \( O(1/\sqrt{t}) \)
+for general convex objectives (\thetaith appropriate diminishing step
+sizes) . Intuitively, GD’s trajectory is smoother and more
+predictable, while SGD’s path oscillates due to noise but costs far
+less per iteration, enabling many more updates in the same time.
+
+Strongly Convex Case
+
+If \( C(\theta) \) is strongly convex and \( L \)-smooth (so GD enjoys linear
+convergence), the gap \( C(\theta_t)-C(\theta^*) \) for GD shrinks as
+
+$$
+C(\theta_t) - C(\theta^* ) \le \Big(1 - \frac{\mu}{L}\Big)^t [C(\theta_0)-C(\theta^*)],
+$$
+
+a geometric (linear) convergence per iteration . Achieving an
+\( \epsilon \)-accurate solution thus takes on the order of
+\( \log(1/\epsilon) \) iterations for GD. However, each GD iteration costs
+\( O(N) \) gradient evaluations. SGD cannot exploit strong convexity to
+obtain a linear rate – instead, with a properly decaying step size
+(e.g. \( \eta_t = \frac{1}{\mu t} \)) or iterate averaging, SGD attains an
+\( O(1/t) \) convergence rate in expectation . For example, one result
+of Moulines and Bach 2011, see https://papers.nips.cc/paper_files/paper/2011/hash/40008b9a5380fcacce3976bf7c08af5b-Abstract.html shows that with \( \eta_t = \Theta(1/t) \),
+
+$$
+\mathbb{E}[C(\theta_t) - C(\theta^*)] = O(1/t),
+$$
+
+for strongly convex, smooth \( F \) . This \( 1/t \) rate is slower per
+iteration than GD’s exponential decay, but each SGD iteration is \( N \)
+times cheaper. In fact, to reach error \( \epsilon \), plain SGD needs on
+the order of \( T=O(1/\epsilon) \) iterations (sub-linear convergence),
+while GD needs \( O(\log(1/\epsilon)) \) iterations. When accounting for
+cost-per-iteration, GD requires \( O(N \log(1/\epsilon)) \) total gradient
+computations versus SGD’s \( O(1/\epsilon) \) single-sample
+computations. In large-scale regimes (huge \( N \)), SGD can be
+faster in wall-clock time because \( N \log(1/\epsilon) \) may far exceed
+\( 1/\epsilon \) for reasonable accuracy levels. In other words,
+with millions of data points, one epoch of GD (one full gradient) is
+extremely costly, whereas SGD can make \( N \) cheap updates in the time
+GD makes one – often yielding a good solution faster in practice, even
+though SGD’s asymptotic error decays more slowly. As one lecture
+succinctly puts it: “SGD can be super effective in terms of iteration
+cost and memory, but SGD is slow to converge and can’t adapt to strong
+convexity” . Thus, the break-even point depends on \( N \) and the desired
+accuracy: for moderate accuracy on very large \( N \), SGD’s cheaper
+updates win; for extremely high precision (very small \( \epsilon \)) on a
+modest \( N \), GD’s fast convergence per step can be advantageous.
+
+Non-Convex Problems
+
+In non-convex optimization (e.g. deep neural networks), neither GD nor
+SGD guarantees global minima, but SGD often displays faster progress
+in finding useful minima. Theoretical results here are weaker, usually
+showing convergence to a stationary point \( \theta \) (\( |\nabla C| \) is
+small) in expectation. For example, GD might require \( O(1/\epsilon^2) \)
+iterations to ensure \( |\nabla C(\theta)| < \epsilon \), and SGD typically has
+similar polynomial complexity (often worse due to gradient
+noise). However, a noteworthy difference is that SGD’s stochasticity
+can help escape saddle points or poor local minima. Random gradient
+fluctuations act like implicit noise, helping the iterate “jump” out
+of flat saddle regions where full-batch GD could stagnate . In fact,
+research has shown that adding noise to GD can guarantee escaping
+saddle points in polynomial time, and the inherent noise in SGD often
+serves this role. Empirically, this means SGD can sometimes find a
+lower loss basin faster, whereas full-batch GD might get “stuck” near
+saddle points or need a very small learning rate to navigate complex
+error surfaces . Overall, in modern high-dimensional machine learning,
+SGD (or mini-batch SGD) is the workhorse for large non-convex problems
+because it converges to good solutions much faster in practice,
+despite the lack of a linear convergence guarantee. Full-batch GD is
+rarely used on large neural networks, as it would require tiny steps
+to avoid divergence and is extremely slow per iteration .
+
+
+
+Memory Usage and Scalability
+
+A major advantage of SGD is its memory efficiency in handling large
+datasets. Full-batch GD requires access to the entire training set for
+each iteration, which often means the whole dataset (or a large
+subset) must reside in memory to compute \( \nabla C(\theta) \) . This results
+in memory usage that scales linearly with the dataset size \( N \). For
+instance, if each training sample is large (e.g. high-dimensional
+features), computing a full gradient may require storing a substantial
+portion of the data or all intermediate gradients until they are
+aggregated. In contrast, SGD needs only a single (or a small
+mini-batch of) training example(s) in memory at any time . The
+algorithm processes one sample (or mini-batch) at a time and
+immediately updates the model, discarding that sample before moving to
+the next. This streaming approach means that memory footprint is
+essentially independent of \( N \) (apart from storing the model
+parameters themselves). As one source notes, gradient descent
+“requires more memory than SGD” because it “must store the entire
+dataset for each iteration,” whereas SGD “only needs to store the
+current training example” . In practical terms, if you have a dataset
+of size, say, 1 million examples, full-batch GD would need memory for
+all million every step, while SGD could be implemented to load just
+one example at a time – a crucial benefit if data are too large to fit
+in RAM or GPU memory. This scalability makes SGD suitable for
+large-scale learning: as long as you can stream data from disk, SGD
+can handle arbitrarily large datasets with fixed memory. In fact, SGD
+“does not need to remember which examples were visited” in the past,
+allowing it to run in an online fashion on infinite data streams
+. Full-batch GD, on the other hand, would require multiple passes
+through a giant dataset per update (or a complex distributed memory
+system), which is often infeasible.
+
+
+There is also a secondary memory effect: computing a full-batch
+gradient in deep learning requires storing all intermediate
+activations for backpropagation across the entire batch. A very large
+batch (approaching the full dataset) might exhaust GPU memory due to
+the need to hold activation gradients for thousands or millions of
+examples simultaneously. SGD/minibatches mitigate this by splitting
+the workload – e.g. with a mini-batch of size 32 or 256, memory use
+stays bounded, whereas a full-batch (size = \( N \)) forward/backward pass
+could not even be executed if \( N \) is huge. Techniques like gradient
+accumulation exist to simulate large-batch GD by summing many
+small-batch gradients – but these still process data in manageable
+chunks to avoid memory overflow. In summary, memory complexity for GD
+grows with \( N \), while for SGD it remains \( O(1) \) w.r.t. dataset size
+(only the model and perhaps a mini-batch reside in memory) . This is a
+key reason why batch GD “does not scale” to very large data and why
+virtually all large-scale machine learning algorithms rely on
+stochastic or mini-batch methods.
+
+
+
+Empirical Evidence: Convergence Time and Memory in Practice
+
+Empirical studies strongly support the theoretical trade-offs
+above. In large-scale machine learning tasks, SGD often converges to a
+good solution much faster in wall-clock time than full-batch GD, and
+it uses far less memory. For example, Bottou & Bousquet (2008)
+analyzed learning time under a fixed computational budget and
+concluded that when data is abundant, it’s better to use a faster
+(even if less precise) optimization method to process more examples in
+the same time . This analysis showed that for large-scale problems,
+processing more data with SGD yields lower error than spending the
+time to do exact (batch) optimization on fewer data . In other words,
+if you have a time budget, it’s often optimal to accept slightly
+slower convergence per step (as with SGD) in exchange for being able
+to use many more training samples in that time. This phenomenon is
+borne out by experiments:
+
+Deep Neural Networks
+
+In modern deep learning, full-batch GD is so slow that it is rarely
+attempted; instead, mini-batch SGD is standard. A recent study
+demonstrated that it is possible to train a ResNet-50 on ImageNet
+using full-batch gradient descent, but it required careful tuning
+(e.g. gradient clipping, tiny learning rates) and vast computational
+resources – and even then, each full-batch update was extremely
+expensive.
+
+
+Using a huge batch
+(closer to full GD) tends to slow down convergence if the learning
+rate is not scaled up, and often encounters optimization difficulties
+(plateaus) that small batches avoid.
+Empirically, small or medium
+batch SGD finds minima in fewer clock hours because it can rapidly
+loop over the data with gradient noise aiding exploration.
+
+Memory constraints
+
+From a memory standpoint, practitioners note that batch GD becomes
+infeasible on large data. For example, if one tried to do full-batch
+training on a dataset that doesn’t fit in RAM or GPU memory, the
+program would resort to heavy disk I/O or simply crash. SGD
+circumvents this by processing mini-batches. Even in cases where data
+does fit in memory, using a full batch can spike memory usage due to
+storing all gradients. One empirical observation is that mini-batch
+training has a “lower, fluctuating usage pattern” of memory, whereas
+full-batch loading “quickly consumes memory (often exceeding limits)”
+. This is especially relevant for graph neural networks or other
+models where a “batch” may include a huge chunk of a graph: full-batch
+gradient computation can exhaust GPU memory, whereas mini-batch
+methods keep memory usage manageable .
+
+
+In summary, SGD converges faster than full-batch GD in terms of actual
+training time for large-scale problems, provided we measure
+convergence as reaching a good-enough solution. Theoretical bounds
+show SGD needs more iterations, but because it performs many more
+updates per unit time (and requires far less memory), it often
+achieves lower loss in a given time frame than GD. Full-batch GD might
+take slightly fewer iterations in theory, but each iteration is so
+costly that it is “slower… especially for large datasets” . Meanwhile,
+memory scaling strongly favors SGD: GD’s memory cost grows with
+dataset size, making it impractical beyond a point, whereas SGD’s
+memory use is modest and mostly constant w.r.t. \( N \) . These
+differences have made SGD (and mini-batch variants) the de facto
+choice for training large machine learning models, from logistic
+regression on millions of examples to deep neural networks with
+billions of parameters. The consensus in both research and practice is
+that for large-scale or high-dimensional tasks, SGD-type methods
+converge quicker per unit of computation and handle memory constraints
+better than standard full-batch gradient descent .
+
+
Second moment of the gradient
diff --git a/doc/pub/week37/html/week37.html b/doc/pub/week37/html/week37.html
index 0dbb55be9..ab0cae504 100644
--- a/doc/pub/week37/html/week37.html
+++ b/doc/pub/week37/html/week37.html
@@ -218,6 +218,26 @@ div.toc p,a {
None,
'code-with-a-number-of-minibatches-which-varies'),
('Replace or not', 2, None, 'replace-or-not'),
+ ('SGD vs Full-Batch GD: Convergence Speed and Memory Comparison',
+ 2,
+ None,
+ 'sgd-vs-full-batch-gd-convergence-speed-and-memory-comparison'),
+ ('Theoretical Convergence Speed and convex optimization',
+ 3,
+ None,
+ 'theoretical-convergence-speed-and-convex-optimization'),
+ ('Strongly Convex Case', 3, None, 'strongly-convex-case'),
+ ('Non-Convex Problems', 3, None, 'non-convex-problems'),
+ ('Memory Usage and Scalability',
+ 2,
+ None,
+ 'memory-usage-and-scalability'),
+ ('Empirical Evidence: Convergence Time and Memory in Practice',
+ 2,
+ None,
+ 'empirical-evidence-convergence-time-and-memory-in-practice'),
+ ('Deep Neural Networks', 3, None, 'deep-neural-networks'),
+ ('Memory constraints', 3, None, 'memory-constraints'),
('Second moment of the gradient',
2,
None,
@@ -1310,6 +1330,240 @@ mini-batches. The discussion
useful.
+
+SGD vs Full-Batch GD: Convergence Speed and Memory Comparison
+Theoretical Convergence Speed and convex optimization
+
+Consider minimizing an empirical cost function
+$$
+C(\theta) =\frac{1}{N}\sum_{i=1}^N l_i(\theta),
+$$
+
+where each \( l_i(\theta) \) is a
+differentiable loss term. Gradient Descent (GD) updates parameters
+using the full gradient \( \nabla C(\theta) \), while Stochastic Gradient
+Descent (SGD) uses a single sample (or mini-batch) gradient \( \nabla
+l_i(\theta) \) selected at random. In equation form, one GD step is:
+
+
+$$
+\theta_{t+1} = \theta_t-\eta \nabla C(\theta_t) =\theta_t -\eta \frac{1}{N}\sum_{i=1}^N \nabla l_i(\theta_t),
+$$
+
+whereas one SGD step is:
+
+$$
+\theta_{t+1} = \theta_t -\eta \nabla l_{i_t}(\theta_t),
+$$
+
+with \( i_t \) randomly chosen. On smooth convex problems, GD and SGD both
+converge to the global minimum, but their rates differ. GD can take
+larger, more stable steps since it uses the exact gradient, achieving
+an error that decreases on the order of \( O(1/t) \) per iteration for
+convex objectives (and even exponentially fast for strongly convex
+cases). In contrast, plain SGD has more variance in each step, leading
+to sublinear convergence in expectation – typically \( O(1/\sqrt{t}) \)
+for general convex objectives (\thetaith appropriate diminishing step
+sizes) . Intuitively, GD’s trajectory is smoother and more
+predictable, while SGD’s path oscillates due to noise but costs far
+less per iteration, enabling many more updates in the same time.
+
+Strongly Convex Case
+
+If \( C(\theta) \) is strongly convex and \( L \)-smooth (so GD enjoys linear
+convergence), the gap \( C(\theta_t)-C(\theta^*) \) for GD shrinks as
+
+$$
+C(\theta_t) - C(\theta^* ) \le \Big(1 - \frac{\mu}{L}\Big)^t [C(\theta_0)-C(\theta^*)],
+$$
+
+a geometric (linear) convergence per iteration . Achieving an
+\( \epsilon \)-accurate solution thus takes on the order of
+\( \log(1/\epsilon) \) iterations for GD. However, each GD iteration costs
+\( O(N) \) gradient evaluations. SGD cannot exploit strong convexity to
+obtain a linear rate – instead, with a properly decaying step size
+(e.g. \( \eta_t = \frac{1}{\mu t} \)) or iterate averaging, SGD attains an
+\( O(1/t) \) convergence rate in expectation . For example, one result
+of Moulines and Bach 2011, see https://papers.nips.cc/paper_files/paper/2011/hash/40008b9a5380fcacce3976bf7c08af5b-Abstract.html shows that with \( \eta_t = \Theta(1/t) \),
+
+$$
+\mathbb{E}[C(\theta_t) - C(\theta^*)] = O(1/t),
+$$
+
+for strongly convex, smooth \( F \) . This \( 1/t \) rate is slower per
+iteration than GD’s exponential decay, but each SGD iteration is \( N \)
+times cheaper. In fact, to reach error \( \epsilon \), plain SGD needs on
+the order of \( T=O(1/\epsilon) \) iterations (sub-linear convergence),
+while GD needs \( O(\log(1/\epsilon)) \) iterations. When accounting for
+cost-per-iteration, GD requires \( O(N \log(1/\epsilon)) \) total gradient
+computations versus SGD’s \( O(1/\epsilon) \) single-sample
+computations. In large-scale regimes (huge \( N \)), SGD can be
+faster in wall-clock time because \( N \log(1/\epsilon) \) may far exceed
+\( 1/\epsilon \) for reasonable accuracy levels. In other words,
+with millions of data points, one epoch of GD (one full gradient) is
+extremely costly, whereas SGD can make \( N \) cheap updates in the time
+GD makes one – often yielding a good solution faster in practice, even
+though SGD’s asymptotic error decays more slowly. As one lecture
+succinctly puts it: “SGD can be super effective in terms of iteration
+cost and memory, but SGD is slow to converge and can’t adapt to strong
+convexity” . Thus, the break-even point depends on \( N \) and the desired
+accuracy: for moderate accuracy on very large \( N \), SGD’s cheaper
+updates win; for extremely high precision (very small \( \epsilon \)) on a
+modest \( N \), GD’s fast convergence per step can be advantageous.
+
+Non-Convex Problems
+
+In non-convex optimization (e.g. deep neural networks), neither GD nor
+SGD guarantees global minima, but SGD often displays faster progress
+in finding useful minima. Theoretical results here are weaker, usually
+showing convergence to a stationary point \( \theta \) (\( |\nabla C| \) is
+small) in expectation. For example, GD might require \( O(1/\epsilon^2) \)
+iterations to ensure \( |\nabla C(\theta)| < \epsilon \), and SGD typically has
+similar polynomial complexity (often worse due to gradient
+noise). However, a noteworthy difference is that SGD’s stochasticity
+can help escape saddle points or poor local minima. Random gradient
+fluctuations act like implicit noise, helping the iterate “jump” out
+of flat saddle regions where full-batch GD could stagnate . In fact,
+research has shown that adding noise to GD can guarantee escaping
+saddle points in polynomial time, and the inherent noise in SGD often
+serves this role. Empirically, this means SGD can sometimes find a
+lower loss basin faster, whereas full-batch GD might get “stuck” near
+saddle points or need a very small learning rate to navigate complex
+error surfaces . Overall, in modern high-dimensional machine learning,
+SGD (or mini-batch SGD) is the workhorse for large non-convex problems
+because it converges to good solutions much faster in practice,
+despite the lack of a linear convergence guarantee. Full-batch GD is
+rarely used on large neural networks, as it would require tiny steps
+to avoid divergence and is extremely slow per iteration .
+
+
+
+Memory Usage and Scalability
+
+A major advantage of SGD is its memory efficiency in handling large
+datasets. Full-batch GD requires access to the entire training set for
+each iteration, which often means the whole dataset (or a large
+subset) must reside in memory to compute \( \nabla C(\theta) \) . This results
+in memory usage that scales linearly with the dataset size \( N \). For
+instance, if each training sample is large (e.g. high-dimensional
+features), computing a full gradient may require storing a substantial
+portion of the data or all intermediate gradients until they are
+aggregated. In contrast, SGD needs only a single (or a small
+mini-batch of) training example(s) in memory at any time . The
+algorithm processes one sample (or mini-batch) at a time and
+immediately updates the model, discarding that sample before moving to
+the next. This streaming approach means that memory footprint is
+essentially independent of \( N \) (apart from storing the model
+parameters themselves). As one source notes, gradient descent
+“requires more memory than SGD” because it “must store the entire
+dataset for each iteration,” whereas SGD “only needs to store the
+current training example” . In practical terms, if you have a dataset
+of size, say, 1 million examples, full-batch GD would need memory for
+all million every step, while SGD could be implemented to load just
+one example at a time – a crucial benefit if data are too large to fit
+in RAM or GPU memory. This scalability makes SGD suitable for
+large-scale learning: as long as you can stream data from disk, SGD
+can handle arbitrarily large datasets with fixed memory. In fact, SGD
+“does not need to remember which examples were visited” in the past,
+allowing it to run in an online fashion on infinite data streams
+. Full-batch GD, on the other hand, would require multiple passes
+through a giant dataset per update (or a complex distributed memory
+system), which is often infeasible.
+
+
+There is also a secondary memory effect: computing a full-batch
+gradient in deep learning requires storing all intermediate
+activations for backpropagation across the entire batch. A very large
+batch (approaching the full dataset) might exhaust GPU memory due to
+the need to hold activation gradients for thousands or millions of
+examples simultaneously. SGD/minibatches mitigate this by splitting
+the workload – e.g. with a mini-batch of size 32 or 256, memory use
+stays bounded, whereas a full-batch (size = \( N \)) forward/backward pass
+could not even be executed if \( N \) is huge. Techniques like gradient
+accumulation exist to simulate large-batch GD by summing many
+small-batch gradients – but these still process data in manageable
+chunks to avoid memory overflow. In summary, memory complexity for GD
+grows with \( N \), while for SGD it remains \( O(1) \) w.r.t. dataset size
+(only the model and perhaps a mini-batch reside in memory) . This is a
+key reason why batch GD “does not scale” to very large data and why
+virtually all large-scale machine learning algorithms rely on
+stochastic or mini-batch methods.
+
+
+
+Empirical Evidence: Convergence Time and Memory in Practice
+
+Empirical studies strongly support the theoretical trade-offs
+above. In large-scale machine learning tasks, SGD often converges to a
+good solution much faster in wall-clock time than full-batch GD, and
+it uses far less memory. For example, Bottou & Bousquet (2008)
+analyzed learning time under a fixed computational budget and
+concluded that when data is abundant, it’s better to use a faster
+(even if less precise) optimization method to process more examples in
+the same time . This analysis showed that for large-scale problems,
+processing more data with SGD yields lower error than spending the
+time to do exact (batch) optimization on fewer data . In other words,
+if you have a time budget, it’s often optimal to accept slightly
+slower convergence per step (as with SGD) in exchange for being able
+to use many more training samples in that time. This phenomenon is
+borne out by experiments:
+
+Deep Neural Networks
+
+In modern deep learning, full-batch GD is so slow that it is rarely
+attempted; instead, mini-batch SGD is standard. A recent study
+demonstrated that it is possible to train a ResNet-50 on ImageNet
+using full-batch gradient descent, but it required careful tuning
+(e.g. gradient clipping, tiny learning rates) and vast computational
+resources – and even then, each full-batch update was extremely
+expensive.
+
+
+Using a huge batch
+(closer to full GD) tends to slow down convergence if the learning
+rate is not scaled up, and often encounters optimization difficulties
+(plateaus) that small batches avoid.
+Empirically, small or medium
+batch SGD finds minima in fewer clock hours because it can rapidly
+loop over the data with gradient noise aiding exploration.
+
+Memory constraints
+
+From a memory standpoint, practitioners note that batch GD becomes
+infeasible on large data. For example, if one tried to do full-batch
+training on a dataset that doesn’t fit in RAM or GPU memory, the
+program would resort to heavy disk I/O or simply crash. SGD
+circumvents this by processing mini-batches. Even in cases where data
+does fit in memory, using a full batch can spike memory usage due to
+storing all gradients. One empirical observation is that mini-batch
+training has a “lower, fluctuating usage pattern” of memory, whereas
+full-batch loading “quickly consumes memory (often exceeding limits)”
+. This is especially relevant for graph neural networks or other
+models where a “batch” may include a huge chunk of a graph: full-batch
+gradient computation can exhaust GPU memory, whereas mini-batch
+methods keep memory usage manageable .
+
+
+In summary, SGD converges faster than full-batch GD in terms of actual
+training time for large-scale problems, provided we measure
+convergence as reaching a good-enough solution. Theoretical bounds
+show SGD needs more iterations, but because it performs many more
+updates per unit time (and requires far less memory), it often
+achieves lower loss in a given time frame than GD. Full-batch GD might
+take slightly fewer iterations in theory, but each iteration is so
+costly that it is “slower… especially for large datasets” . Meanwhile,
+memory scaling strongly favors SGD: GD’s memory cost grows with
+dataset size, making it impractical beyond a point, whereas SGD’s
+memory use is modest and mostly constant w.r.t. \( N \) . These
+differences have made SGD (and mini-batch variants) the de facto
+choice for training large machine learning models, from logistic
+regression on millions of examples to deep neural networks with
+billions of parameters. The consensus in both research and practice is
+that for large-scale or high-dimensional tasks, SGD-type methods
+converge quicker per unit of computation and handle memory constraints
+better than standard full-batch gradient descent .
+
+
Second moment of the gradient
diff --git a/doc/pub/week37/ipynb/ipynb-week37-src.tar.gz b/doc/pub/week37/ipynb/ipynb-week37-src.tar.gz
index 42c3bbc146048d2cc062b4e053c3da9ddeecfe2b..3b29921efd3bec48e44352780778e7e1ffbdd5a3 100644
GIT binary patch
delta 39
ucmdncF1Mjwj$OW+gCQq>Un6@fJ7X(5Q!6`jD?3XoJ8LUD+g5h=mCXR^P72Wg
delta 39
vcmdncF1Mjwj$OW+gJH$9y^ZXx?2N7KOs(w9t?VqV?5wTqY+KpcS2hCx{ZR{p
diff --git a/doc/pub/week37/ipynb/week37.ipynb b/doc/pub/week37/ipynb/week37.ipynb
index 9038066ab..8e275c56f 100644
--- a/doc/pub/week37/ipynb/week37.ipynb
+++ b/doc/pub/week37/ipynb/week37.ipynb
@@ -2,7 +2,7 @@
"cells": [
{
"cell_type": "markdown",
- "id": "311a2385",
+ "id": "53d0b4e7",
"metadata": {
"editable": true
},
@@ -14,7 +14,7 @@
},
{
"cell_type": "markdown",
- "id": "9e4484dc",
+ "id": "0c919844",
"metadata": {
"editable": true
},
@@ -29,7 +29,7 @@
},
{
"cell_type": "markdown",
- "id": "a24010ae",
+ "id": "a5769b3d",
"metadata": {
"editable": true
},
@@ -52,7 +52,7 @@
},
{
"cell_type": "markdown",
- "id": "4a291d59",
+ "id": "21f2937f",
"metadata": {
"editable": true
},
@@ -69,7 +69,7 @@
},
{
"cell_type": "markdown",
- "id": "85c747e2",
+ "id": "da32a24e",
"metadata": {
"editable": true
},
@@ -79,7 +79,7 @@
},
{
"cell_type": "markdown",
- "id": "6580dfe2",
+ "id": "2b7f8433",
"metadata": {
"editable": true
},
@@ -103,7 +103,7 @@
{
"cell_type": "code",
"execution_count": 1,
- "id": "c2ddcfe5",
+ "id": "2a8c5baa",
"metadata": {
"collapsed": false,
"editable": true
@@ -117,7 +117,7 @@
},
{
"cell_type": "markdown",
- "id": "e1e8a5b2",
+ "id": "47d8423d",
"metadata": {
"editable": true
},
@@ -128,7 +128,7 @@
},
{
"cell_type": "markdown",
- "id": "c8a5100b",
+ "id": "08d37b91",
"metadata": {
"editable": true
},
@@ -140,7 +140,7 @@
},
{
"cell_type": "markdown",
- "id": "b026883e",
+ "id": "a3c412ef",
"metadata": {
"editable": true
},
@@ -150,7 +150,7 @@
},
{
"cell_type": "markdown",
- "id": "3a2f7b75",
+ "id": "3f1b2071",
"metadata": {
"editable": true
},
@@ -162,7 +162,7 @@
},
{
"cell_type": "markdown",
- "id": "6380eed5",
+ "id": "07536cbd",
"metadata": {
"editable": true
},
@@ -176,7 +176,7 @@
},
{
"cell_type": "markdown",
- "id": "c5d3766d",
+ "id": "34287135",
"metadata": {
"editable": true
},
@@ -192,7 +192,7 @@
},
{
"cell_type": "markdown",
- "id": "1d313807",
+ "id": "c1063bfb",
"metadata": {
"editable": true
},
@@ -202,7 +202,7 @@
},
{
"cell_type": "markdown",
- "id": "bee64882",
+ "id": "d327d2e3",
"metadata": {
"editable": true
},
@@ -214,7 +214,7 @@
},
{
"cell_type": "markdown",
- "id": "7ffe8d02",
+ "id": "b2c9a9cd",
"metadata": {
"editable": true
},
@@ -224,7 +224,7 @@
},
{
"cell_type": "markdown",
- "id": "97225362",
+ "id": "062a7534",
"metadata": {
"editable": true
},
@@ -236,7 +236,7 @@
},
{
"cell_type": "markdown",
- "id": "9fe2a0b3",
+ "id": "b1b15536",
"metadata": {
"editable": true
},
@@ -250,7 +250,7 @@
},
{
"cell_type": "markdown",
- "id": "2e678439",
+ "id": "a259e250",
"metadata": {
"editable": true
},
@@ -260,7 +260,7 @@
},
{
"cell_type": "markdown",
- "id": "5f45e358",
+ "id": "002197c1",
"metadata": {
"editable": true
},
@@ -271,7 +271,7 @@
},
{
"cell_type": "markdown",
- "id": "1713ee43",
+ "id": "55e8dca9",
"metadata": {
"editable": true
},
@@ -286,7 +286,7 @@
},
{
"cell_type": "markdown",
- "id": "671ea0fc",
+ "id": "a97ffaec",
"metadata": {
"editable": true
},
@@ -296,7 +296,7 @@
},
{
"cell_type": "markdown",
- "id": "7df56d17",
+ "id": "b59a4220",
"metadata": {
"editable": true
},
@@ -308,7 +308,7 @@
},
{
"cell_type": "markdown",
- "id": "5887c657",
+ "id": "ba5dcc08",
"metadata": {
"editable": true
},
@@ -320,7 +320,7 @@
},
{
"cell_type": "markdown",
- "id": "5a012ac0",
+ "id": "d8907fed",
"metadata": {
"editable": true
},
@@ -335,7 +335,7 @@
},
{
"cell_type": "markdown",
- "id": "cf1fd4f4",
+ "id": "728d5b78",
"metadata": {
"editable": true
},
@@ -348,7 +348,7 @@
{
"cell_type": "code",
"execution_count": 2,
- "id": "4417d3aa",
+ "id": "02d7e401",
"metadata": {
"collapsed": false,
"editable": true
@@ -407,7 +407,7 @@
},
{
"cell_type": "markdown",
- "id": "7d39d005",
+ "id": "4dd147c2",
"metadata": {
"editable": true
},
@@ -419,7 +419,7 @@
},
{
"cell_type": "markdown",
- "id": "45a85d32",
+ "id": "75ca7f80",
"metadata": {
"editable": true
},
@@ -431,7 +431,7 @@
},
{
"cell_type": "markdown",
- "id": "31d267ea",
+ "id": "5b897c75",
"metadata": {
"editable": true
},
@@ -441,7 +441,7 @@
},
{
"cell_type": "markdown",
- "id": "f8f50b02",
+ "id": "46aa12f6",
"metadata": {
"editable": true
},
@@ -455,7 +455,7 @@
},
{
"cell_type": "markdown",
- "id": "ac21d44c",
+ "id": "8ac05816",
"metadata": {
"editable": true
},
@@ -465,7 +465,7 @@
},
{
"cell_type": "markdown",
- "id": "aae5aaa1",
+ "id": "cee76d94",
"metadata": {
"editable": true
},
@@ -477,7 +477,7 @@
},
{
"cell_type": "markdown",
- "id": "319922a5",
+ "id": "88cf9577",
"metadata": {
"editable": true
},
@@ -488,7 +488,7 @@
},
{
"cell_type": "markdown",
- "id": "724078a1",
+ "id": "0108d67e",
"metadata": {
"editable": true
},
@@ -503,7 +503,7 @@
},
{
"cell_type": "markdown",
- "id": "dbc443e3",
+ "id": "1e307469",
"metadata": {
"editable": true
},
@@ -517,7 +517,7 @@
},
{
"cell_type": "markdown",
- "id": "2ea2bf50",
+ "id": "c8dc7485",
"metadata": {
"editable": true
},
@@ -528,7 +528,7 @@
{
"cell_type": "code",
"execution_count": 3,
- "id": "9f431da1",
+ "id": "2909407a",
"metadata": {
"collapsed": false,
"editable": true
@@ -589,7 +589,7 @@
},
{
"cell_type": "markdown",
- "id": "8aa155a9",
+ "id": "d25693ff",
"metadata": {
"editable": true
},
@@ -611,7 +611,7 @@
},
{
"cell_type": "markdown",
- "id": "03bd2e44",
+ "id": "78b0bf65",
"metadata": {
"editable": true
},
@@ -626,7 +626,7 @@
},
{
"cell_type": "markdown",
- "id": "0e101e2d",
+ "id": "9ee803d8",
"metadata": {
"editable": true
},
@@ -637,7 +637,7 @@
{
"cell_type": "code",
"execution_count": 4,
- "id": "09ecede4",
+ "id": "ac420f7a",
"metadata": {
"collapsed": false,
"editable": true
@@ -703,7 +703,7 @@
},
{
"cell_type": "markdown",
- "id": "3489dbbc",
+ "id": "c548d574",
"metadata": {
"editable": true
},
@@ -714,7 +714,7 @@
{
"cell_type": "code",
"execution_count": 5,
- "id": "426eaa39",
+ "id": "687e9d89",
"metadata": {
"collapsed": false,
"editable": true
@@ -788,7 +788,7 @@
},
{
"cell_type": "markdown",
- "id": "6220214d",
+ "id": "27a27a67",
"metadata": {
"editable": true
},
@@ -807,7 +807,7 @@
},
{
"cell_type": "markdown",
- "id": "bf86ac65",
+ "id": "a12c19b2",
"metadata": {
"editable": true
},
@@ -828,7 +828,7 @@
},
{
"cell_type": "markdown",
- "id": "4ac61edb",
+ "id": "b8436434",
"metadata": {
"editable": true
},
@@ -844,7 +844,7 @@
},
{
"cell_type": "markdown",
- "id": "0058008d",
+ "id": "f00e1864",
"metadata": {
"editable": true
},
@@ -858,7 +858,7 @@
},
{
"cell_type": "markdown",
- "id": "f994e1e2",
+ "id": "ea91af30",
"metadata": {
"editable": true
},
@@ -886,7 +886,7 @@
},
{
"cell_type": "markdown",
- "id": "842a8611",
+ "id": "bc9502a0",
"metadata": {
"editable": true
},
@@ -918,7 +918,7 @@
},
{
"cell_type": "markdown",
- "id": "90bd121a",
+ "id": "6a236a2a",
"metadata": {
"editable": true
},
@@ -935,7 +935,7 @@
},
{
"cell_type": "markdown",
- "id": "5cd81303",
+ "id": "29dc562b",
"metadata": {
"editable": true
},
@@ -948,7 +948,7 @@
},
{
"cell_type": "markdown",
- "id": "60e085a9",
+ "id": "6a34f155",
"metadata": {
"editable": true
},
@@ -961,7 +961,7 @@
},
{
"cell_type": "markdown",
- "id": "fef0100e",
+ "id": "0afd8cd8",
"metadata": {
"editable": true
},
@@ -974,7 +974,7 @@
},
{
"cell_type": "markdown",
- "id": "aaba7f05",
+ "id": "f0b27e71",
"metadata": {
"editable": true
},
@@ -988,7 +988,7 @@
},
{
"cell_type": "markdown",
- "id": "038b47ae",
+ "id": "3b04b9c6",
"metadata": {
"editable": true
},
@@ -1010,7 +1010,7 @@
},
{
"cell_type": "markdown",
- "id": "0ad42833",
+ "id": "05eca708",
"metadata": {
"editable": true
},
@@ -1025,7 +1025,7 @@
},
{
"cell_type": "markdown",
- "id": "64b15ba2",
+ "id": "473025f4",
"metadata": {
"editable": true
},
@@ -1037,7 +1037,7 @@
},
{
"cell_type": "markdown",
- "id": "49c6adb0",
+ "id": "26e0b288",
"metadata": {
"editable": true
},
@@ -1050,7 +1050,7 @@
},
{
"cell_type": "markdown",
- "id": "82873545",
+ "id": "091efee5",
"metadata": {
"editable": true
},
@@ -1064,7 +1064,7 @@
},
{
"cell_type": "markdown",
- "id": "35a8e70d",
+ "id": "22c5f80e",
"metadata": {
"editable": true
},
@@ -1075,7 +1075,7 @@
{
"cell_type": "code",
"execution_count": 6,
- "id": "6aa32b90",
+ "id": "102b1658",
"metadata": {
"collapsed": false,
"editable": true
@@ -1100,7 +1100,7 @@
},
{
"cell_type": "markdown",
- "id": "6e20f534",
+ "id": "79448e46",
"metadata": {
"editable": true
},
@@ -1116,7 +1116,7 @@
},
{
"cell_type": "markdown",
- "id": "71745d3e",
+ "id": "dbc8b940",
"metadata": {
"editable": true
},
@@ -1137,7 +1137,7 @@
},
{
"cell_type": "markdown",
- "id": "bad95be2",
+ "id": "b63ae18d",
"metadata": {
"editable": true
},
@@ -1157,7 +1157,7 @@
},
{
"cell_type": "markdown",
- "id": "40b4d87e",
+ "id": "c5ee074e",
"metadata": {
"editable": true
},
@@ -1176,7 +1176,7 @@
{
"cell_type": "code",
"execution_count": 7,
- "id": "1208bbec",
+ "id": "cfc48413",
"metadata": {
"collapsed": false,
"editable": true
@@ -1211,7 +1211,7 @@
},
{
"cell_type": "markdown",
- "id": "b83b5ed1",
+ "id": "fbb8c0eb",
"metadata": {
"editable": true
},
@@ -1224,7 +1224,7 @@
{
"cell_type": "code",
"execution_count": 8,
- "id": "1f669db6",
+ "id": "cc1e51cd",
"metadata": {
"collapsed": false,
"editable": true
@@ -1301,7 +1301,7 @@
},
{
"cell_type": "markdown",
- "id": "3e9ed564",
+ "id": "22a23ea0",
"metadata": {
"editable": true
},
@@ -1316,7 +1316,377 @@
},
{
"cell_type": "markdown",
- "id": "9c0ac318",
+ "id": "f0258497",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "## SGD vs Full-Batch GD: Convergence Speed and Memory Comparison"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "69f1d941",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "### Theoretical Convergence Speed and convex optimization\n",
+ "\n",
+ "Consider minimizing an empirical cost function"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "876e1d2b",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "$$\n",
+ "C(\\theta) =\\frac{1}{N}\\sum_{i=1}^N l_i(\\theta),\n",
+ "$$"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "67381fd3",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "where each $l_i(\\theta)$ is a\n",
+ "differentiable loss term. Gradient Descent (GD) updates parameters\n",
+ "using the full gradient $\\nabla C(\\theta)$, while Stochastic Gradient\n",
+ "Descent (SGD) uses a single sample (or mini-batch) gradient $\\nabla\n",
+ "l_i(\\theta)$ selected at random. In equation form, one GD step is:"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "f08ec530",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "$$\n",
+ "\\theta_{t+1} = \\theta_t-\\eta \\nabla C(\\theta_t) =\\theta_t -\\eta \\frac{1}{N}\\sum_{i=1}^N \\nabla l_i(\\theta_t),\n",
+ "$$"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "3d99c5ee",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "whereas one SGD step is:"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "03ecb7f9",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "$$\n",
+ "\\theta_{t+1} = \\theta_t -\\eta \\nabla l_{i_t}(\\theta_t),\n",
+ "$$"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "a6db0c6c",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "with $i_t$ randomly chosen. On smooth convex problems, GD and SGD both\n",
+ "converge to the global minimum, but their rates differ. GD can take\n",
+ "larger, more stable steps since it uses the exact gradient, achieving\n",
+ "an error that decreases on the order of $O(1/t)$ per iteration for\n",
+ "convex objectives (and even exponentially fast for strongly convex\n",
+ "cases). In contrast, plain SGD has more variance in each step, leading\n",
+ "to sublinear convergence in expectation – typically $O(1/\\sqrt{t})$\n",
+ "for general convex objectives (\\thetaith appropriate diminishing step\n",
+ "sizes) . Intuitively, GD’s trajectory is smoother and more\n",
+ "predictable, while SGD’s path oscillates due to noise but costs far\n",
+ "less per iteration, enabling many more updates in the same time."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "167f76aa",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "### Strongly Convex Case\n",
+ "\n",
+ "If $C(\\theta)$ is strongly convex and $L$-smooth (so GD enjoys linear\n",
+ "convergence), the gap $C(\\theta_t)-C(\\theta^*)$ for GD shrinks as"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "d8d5cb23",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "$$\n",
+ "C(\\theta_t) - C(\\theta^* ) \\le \\Big(1 - \\frac{\\mu}{L}\\Big)^t [C(\\theta_0)-C(\\theta^*)],\n",
+ "$$"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "e127c141",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "a geometric (linear) convergence per iteration . Achieving an\n",
+ "$\\epsilon$-accurate solution thus takes on the order of\n",
+ "$\\log(1/\\epsilon)$ iterations for GD. However, each GD iteration costs\n",
+ "$O(N)$ gradient evaluations. SGD cannot exploit strong convexity to\n",
+ "obtain a linear rate – instead, with a properly decaying step size\n",
+ "(e.g. $\\eta_t = \\frac{1}{\\mu t}$) or iterate averaging, SGD attains an\n",
+ "$O(1/t)$ convergence rate in expectation . For example, one result\n",
+ "of Moulines and Bach 2011, see shows that with $\\eta_t = \\Theta(1/t)$,"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "7ea2c03c",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "$$\n",
+ "\\mathbb{E}[C(\\theta_t) - C(\\theta^*)] = O(1/t),\n",
+ "$$"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "e2e33c54",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "for strongly convex, smooth $F$ . This $1/t$ rate is slower per\n",
+ "iteration than GD’s exponential decay, but each SGD iteration is $N$\n",
+ "times cheaper. In fact, to reach error $\\epsilon$, plain SGD needs on\n",
+ "the order of $T=O(1/\\epsilon)$ iterations (sub-linear convergence),\n",
+ "while GD needs $O(\\log(1/\\epsilon))$ iterations. When accounting for\n",
+ "cost-per-iteration, GD requires $O(N \\log(1/\\epsilon))$ total gradient\n",
+ "computations versus SGD’s $O(1/\\epsilon)$ single-sample\n",
+ "computations. In large-scale regimes (huge $N$), SGD can be\n",
+ "faster in wall-clock time because $N \\log(1/\\epsilon)$ may far exceed\n",
+ "$1/\\epsilon$ for reasonable accuracy levels. In other words,\n",
+ "with millions of data points, one epoch of GD (one full gradient) is\n",
+ "extremely costly, whereas SGD can make $N$ cheap updates in the time\n",
+ "GD makes one – often yielding a good solution faster in practice, even\n",
+ "though SGD’s asymptotic error decays more slowly. As one lecture\n",
+ "succinctly puts it: “SGD can be super effective in terms of iteration\n",
+ "cost and memory, but SGD is slow to converge and can’t adapt to strong\n",
+ "convexity” . Thus, the break-even point depends on $N$ and the desired\n",
+ "accuracy: for moderate accuracy on very large $N$, SGD’s cheaper\n",
+ "updates win; for extremely high precision (very small $\\epsilon$) on a\n",
+ "modest $N$, GD’s fast convergence per step can be advantageous."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "88f943e6",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "### Non-Convex Problems\n",
+ "\n",
+ "In non-convex optimization (e.g. deep neural networks), neither GD nor\n",
+ "SGD guarantees global minima, but SGD often displays faster progress\n",
+ "in finding useful minima. Theoretical results here are weaker, usually\n",
+ "showing convergence to a stationary point $\\theta$ ($|\\nabla C|$ is\n",
+ "small) in expectation. For example, GD might require $O(1/\\epsilon^2)$\n",
+ "iterations to ensure $|\\nabla C(\\theta)| < \\epsilon$, and SGD typically has\n",
+ "similar polynomial complexity (often worse due to gradient\n",
+ "noise). However, a noteworthy difference is that SGD’s stochasticity\n",
+ "can help escape saddle points or poor local minima. Random gradient\n",
+ "fluctuations act like implicit noise, helping the iterate “jump” out\n",
+ "of flat saddle regions where full-batch GD could stagnate . In fact,\n",
+ "research has shown that adding noise to GD can guarantee escaping\n",
+ "saddle points in polynomial time, and the inherent noise in SGD often\n",
+ "serves this role. Empirically, this means SGD can sometimes find a\n",
+ "lower loss basin faster, whereas full-batch GD might get “stuck” near\n",
+ "saddle points or need a very small learning rate to navigate complex\n",
+ "error surfaces . Overall, in modern high-dimensional machine learning,\n",
+ "SGD (or mini-batch SGD) is the workhorse for large non-convex problems\n",
+ "because it converges to good solutions much faster in practice,\n",
+ "despite the lack of a linear convergence guarantee. Full-batch GD is\n",
+ "rarely used on large neural networks, as it would require tiny steps\n",
+ "to avoid divergence and is extremely slow per iteration ."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "aa4a8927",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "## Memory Usage and Scalability\n",
+ "\n",
+ "A major advantage of SGD is its memory efficiency in handling large\n",
+ "datasets. Full-batch GD requires access to the entire training set for\n",
+ "each iteration, which often means the whole dataset (or a large\n",
+ "subset) must reside in memory to compute $\\nabla C(\\theta)$ . This results\n",
+ "in memory usage that scales linearly with the dataset size $N$. For\n",
+ "instance, if each training sample is large (e.g. high-dimensional\n",
+ "features), computing a full gradient may require storing a substantial\n",
+ "portion of the data or all intermediate gradients until they are\n",
+ "aggregated. In contrast, SGD needs only a single (or a small\n",
+ "mini-batch of) training example(s) in memory at any time . The\n",
+ "algorithm processes one sample (or mini-batch) at a time and\n",
+ "immediately updates the model, discarding that sample before moving to\n",
+ "the next. This streaming approach means that memory footprint is\n",
+ "essentially independent of $N$ (apart from storing the model\n",
+ "parameters themselves). As one source notes, gradient descent\n",
+ "“requires more memory than SGD” because it “must store the entire\n",
+ "dataset for each iteration,” whereas SGD “only needs to store the\n",
+ "current training example” . In practical terms, if you have a dataset\n",
+ "of size, say, 1 million examples, full-batch GD would need memory for\n",
+ "all million every step, while SGD could be implemented to load just\n",
+ "one example at a time – a crucial benefit if data are too large to fit\n",
+ "in RAM or GPU memory. This scalability makes SGD suitable for\n",
+ "large-scale learning: as long as you can stream data from disk, SGD\n",
+ "can handle arbitrarily large datasets with fixed memory. In fact, SGD\n",
+ "“does not need to remember which examples were visited” in the past,\n",
+ "allowing it to run in an online fashion on infinite data streams\n",
+ ". Full-batch GD, on the other hand, would require multiple passes\n",
+ "through a giant dataset per update (or a complex distributed memory\n",
+ "system), which is often infeasible.\n",
+ "\n",
+ "There is also a secondary memory effect: computing a full-batch\n",
+ "gradient in deep learning requires storing all intermediate\n",
+ "activations for backpropagation across the entire batch. A very large\n",
+ "batch (approaching the full dataset) might exhaust GPU memory due to\n",
+ "the need to hold activation gradients for thousands or millions of\n",
+ "examples simultaneously. SGD/minibatches mitigate this by splitting\n",
+ "the workload – e.g. with a mini-batch of size 32 or 256, memory use\n",
+ "stays bounded, whereas a full-batch (size = $N$) forward/backward pass\n",
+ "could not even be executed if $N$ is huge. Techniques like gradient\n",
+ "accumulation exist to simulate large-batch GD by summing many\n",
+ "small-batch gradients – but these still process data in manageable\n",
+ "chunks to avoid memory overflow. In summary, memory complexity for GD\n",
+ "grows with $N$, while for SGD it remains $O(1)$ w.r.t. dataset size\n",
+ "(only the model and perhaps a mini-batch reside in memory) . This is a\n",
+ "key reason why batch GD “does not scale” to very large data and why\n",
+ "virtually all large-scale machine learning algorithms rely on\n",
+ "stochastic or mini-batch methods."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "72d0192b",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "## Empirical Evidence: Convergence Time and Memory in Practice\n",
+ "\n",
+ "Empirical studies strongly support the theoretical trade-offs\n",
+ "above. In large-scale machine learning tasks, SGD often converges to a\n",
+ "good solution much faster in wall-clock time than full-batch GD, and\n",
+ "it uses far less memory. For example, Bottou & Bousquet (2008)\n",
+ "analyzed learning time under a fixed computational budget and\n",
+ "concluded that when data is abundant, it’s better to use a faster\n",
+ "(even if less precise) optimization method to process more examples in\n",
+ "the same time . This analysis showed that for large-scale problems,\n",
+ "processing more data with SGD yields lower error than spending the\n",
+ "time to do exact (batch) optimization on fewer data . In other words,\n",
+ "if you have a time budget, it’s often optimal to accept slightly\n",
+ "slower convergence per step (as with SGD) in exchange for being able\n",
+ "to use many more training samples in that time. This phenomenon is\n",
+ "borne out by experiments:"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "44fcd423",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "### Deep Neural Networks\n",
+ "\n",
+ "In modern deep learning, full-batch GD is so slow that it is rarely\n",
+ "attempted; instead, mini-batch SGD is standard. A recent study\n",
+ "demonstrated that it is possible to train a ResNet-50 on ImageNet\n",
+ "using full-batch gradient descent, but it required careful tuning\n",
+ "(e.g. gradient clipping, tiny learning rates) and vast computational\n",
+ "resources – and even then, each full-batch update was extremely\n",
+ "expensive.\n",
+ "\n",
+ "Using a huge batch\n",
+ "(closer to full GD) tends to slow down convergence if the learning\n",
+ "rate is not scaled up, and often encounters optimization difficulties\n",
+ "(plateaus) that small batches avoid.\n",
+ "Empirically, small or medium\n",
+ "batch SGD finds minima in fewer clock hours because it can rapidly\n",
+ "loop over the data with gradient noise aiding exploration."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "8de0942f",
+ "metadata": {
+ "editable": true
+ },
+ "source": [
+ "### Memory constraints\n",
+ "\n",
+ "From a memory standpoint, practitioners note that batch GD becomes\n",
+ "infeasible on large data. For example, if one tried to do full-batch\n",
+ "training on a dataset that doesn’t fit in RAM or GPU memory, the\n",
+ "program would resort to heavy disk I/O or simply crash. SGD\n",
+ "circumvents this by processing mini-batches. Even in cases where data\n",
+ "does fit in memory, using a full batch can spike memory usage due to\n",
+ "storing all gradients. One empirical observation is that mini-batch\n",
+ "training has a “lower, fluctuating usage pattern” of memory, whereas\n",
+ "full-batch loading “quickly consumes memory (often exceeding limits)”\n",
+ ". This is especially relevant for graph neural networks or other\n",
+ "models where a “batch” may include a huge chunk of a graph: full-batch\n",
+ "gradient computation can exhaust GPU memory, whereas mini-batch\n",
+ "methods keep memory usage manageable .\n",
+ "\n",
+ "In summary, SGD converges faster than full-batch GD in terms of actual\n",
+ "training time for large-scale problems, provided we measure\n",
+ "convergence as reaching a good-enough solution. Theoretical bounds\n",
+ "show SGD needs more iterations, but because it performs many more\n",
+ "updates per unit time (and requires far less memory), it often\n",
+ "achieves lower loss in a given time frame than GD. Full-batch GD might\n",
+ "take slightly fewer iterations in theory, but each iteration is so\n",
+ "costly that it is “slower… especially for large datasets” . Meanwhile,\n",
+ "memory scaling strongly favors SGD: GD’s memory cost grows with\n",
+ "dataset size, making it impractical beyond a point, whereas SGD’s\n",
+ "memory use is modest and mostly constant w.r.t. $N$ . These\n",
+ "differences have made SGD (and mini-batch variants) the de facto\n",
+ "choice for training large machine learning models, from logistic\n",
+ "regression on millions of examples to deep neural networks with\n",
+ "billions of parameters. The consensus in both research and practice is\n",
+ "that for large-scale or high-dimensional tasks, SGD-type methods\n",
+ "converge quicker per unit of computation and handle memory constraints\n",
+ "better than standard full-batch gradient descent ."
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "f08a4bbe",
"metadata": {
"editable": true
},
@@ -1347,7 +1717,7 @@
},
{
"cell_type": "markdown",
- "id": "d8f518c4",
+ "id": "dc1fa30f",
"metadata": {
"editable": true
},
@@ -1369,7 +1739,7 @@
},
{
"cell_type": "markdown",
- "id": "3dcb89bd",
+ "id": "1fbfcb5e",
"metadata": {
"editable": true
},
@@ -1389,7 +1759,7 @@
},
{
"cell_type": "markdown",
- "id": "8f258bc2",
+ "id": "83d5dfc2",
"metadata": {
"editable": true
},
@@ -1405,7 +1775,7 @@
},
{
"cell_type": "markdown",
- "id": "2a3715f8",
+ "id": "4cf425f2",
"metadata": {
"editable": true
},
@@ -1425,7 +1795,7 @@
},
{
"cell_type": "markdown",
- "id": "a1d9578a",
+ "id": "a8de083c",
"metadata": {
"editable": true
},
@@ -1437,7 +1807,7 @@
},
{
"cell_type": "markdown",
- "id": "b6b5bc5e",
+ "id": "f8b98ecd",
"metadata": {
"editable": true
},
@@ -1449,7 +1819,7 @@
},
{
"cell_type": "markdown",
- "id": "44b313c8",
+ "id": "c41121c9",
"metadata": {
"editable": true
},
@@ -1461,7 +1831,7 @@
},
{
"cell_type": "markdown",
- "id": "b56c85b9",
+ "id": "0c9cde87",
"metadata": {
"editable": true
},
@@ -1473,7 +1843,7 @@
},
{
"cell_type": "markdown",
- "id": "5bcc6bd2",
+ "id": "9079853e",
"metadata": {
"editable": true
},
@@ -1484,7 +1854,7 @@
},
{
"cell_type": "markdown",
- "id": "41fc9f01",
+ "id": "1b2340aa",
"metadata": {
"editable": true
},
@@ -1496,7 +1866,7 @@
},
{
"cell_type": "markdown",
- "id": "8151719b",
+ "id": "1c63eff7",
"metadata": {
"editable": true
},
@@ -1506,7 +1876,7 @@
},
{
"cell_type": "markdown",
- "id": "bb75b0ad",
+ "id": "e05e89e4",
"metadata": {
"editable": true
},
@@ -1518,7 +1888,7 @@
},
{
"cell_type": "markdown",
- "id": "3c71fd46",
+ "id": "b3cbe567",
"metadata": {
"editable": true
},
@@ -1528,7 +1898,7 @@
},
{
"cell_type": "markdown",
- "id": "1d835a18",
+ "id": "5d2f1096",
"metadata": {
"editable": true
},
@@ -1549,7 +1919,7 @@
},
{
"cell_type": "markdown",
- "id": "77dcc8c3",
+ "id": "4c4f3846",
"metadata": {
"editable": true
},
@@ -1562,7 +1932,7 @@
},
{
"cell_type": "markdown",
- "id": "21161d57",
+ "id": "57d24251",
"metadata": {
"editable": true
},
@@ -1574,7 +1944,7 @@
},
{
"cell_type": "markdown",
- "id": "e87e09a9",
+ "id": "caff3ad3",
"metadata": {
"editable": true
},
@@ -1591,7 +1961,7 @@
},
{
"cell_type": "markdown",
- "id": "1a98c681",
+ "id": "67133da8",
"metadata": {
"editable": true
},
@@ -1607,7 +1977,7 @@
},
{
"cell_type": "markdown",
- "id": "8b337277",
+ "id": "d2d2d644",
"metadata": {
"editable": true
},
@@ -1629,7 +1999,7 @@
},
{
"cell_type": "markdown",
- "id": "af77b83f",
+ "id": "897d1ca3",
"metadata": {
"editable": true
},
@@ -1647,7 +2017,7 @@
},
{
"cell_type": "markdown",
- "id": "bc924f77",
+ "id": "549532b3",
"metadata": {
"editable": true
},
@@ -1667,7 +2037,7 @@
},
{
"cell_type": "markdown",
- "id": "86e5ab5e",
+ "id": "f014a3a2",
"metadata": {
"editable": true
},
@@ -1681,7 +2051,7 @@
},
{
"cell_type": "markdown",
- "id": "949f359d",
+ "id": "67bed63f",
"metadata": {
"editable": true
},
@@ -1693,7 +2063,7 @@
},
{
"cell_type": "markdown",
- "id": "0ba26be3",
+ "id": "3014fe59",
"metadata": {
"editable": true
},
@@ -1705,7 +2075,7 @@
},
{
"cell_type": "markdown",
- "id": "4fb9b2a2",
+ "id": "a99d9c1c",
"metadata": {
"editable": true
},
@@ -1717,7 +2087,7 @@
},
{
"cell_type": "markdown",
- "id": "8711e597",
+ "id": "907f9915",
"metadata": {
"editable": true
},
@@ -1729,7 +2099,7 @@
},
{
"cell_type": "markdown",
- "id": "49e6e73d",
+ "id": "551eb7db",
"metadata": {
"editable": true
},
@@ -1740,7 +2110,7 @@
},
{
"cell_type": "markdown",
- "id": "ca5bb491",
+ "id": "ea8ae470",
"metadata": {
"editable": true
},
@@ -1752,7 +2122,7 @@
},
{
"cell_type": "markdown",
- "id": "5e19d7bf",
+ "id": "9f5d78fd",
"metadata": {
"editable": true
},
@@ -1766,7 +2136,7 @@
},
{
"cell_type": "markdown",
- "id": "f79d952e",
+ "id": "8291642f",
"metadata": {
"editable": true
},
@@ -1777,7 +2147,7 @@
},
{
"cell_type": "markdown",
- "id": "13e9862f",
+ "id": "ee0e74ec",
"metadata": {
"editable": true
},
@@ -1789,7 +2159,7 @@
},
{
"cell_type": "markdown",
- "id": "5693500e",
+ "id": "3699f7e5",
"metadata": {
"editable": true
},
@@ -1811,7 +2181,7 @@
},
{
"cell_type": "markdown",
- "id": "65a5e1e7",
+ "id": "16cdd781",
"metadata": {
"editable": true
},
@@ -1835,7 +2205,7 @@
},
{
"cell_type": "markdown",
- "id": "27686255",
+ "id": "779881a9",
"metadata": {
"editable": true
},
@@ -1851,7 +2221,7 @@
},
{
"cell_type": "markdown",
- "id": "f3dfc1e2",
+ "id": "0724c747",
"metadata": {
"editable": true
},
@@ -1867,7 +2237,7 @@
},
{
"cell_type": "markdown",
- "id": "045d399c",
+ "id": "02a113cb",
"metadata": {
"editable": true
},
@@ -1881,7 +2251,7 @@
},
{
"cell_type": "markdown",
- "id": "4e75ee41",
+ "id": "e013ce1a",
"metadata": {
"editable": true
},
@@ -1899,7 +2269,7 @@
},
{
"cell_type": "markdown",
- "id": "ddbb28ab",
+ "id": "1baa5b4e",
"metadata": {
"editable": true
},
@@ -1920,7 +2290,7 @@
{
"cell_type": "code",
"execution_count": 9,
- "id": "dae38b6c",
+ "id": "8aab54bd",
"metadata": {
"collapsed": false,
"editable": true
@@ -1980,7 +2350,7 @@
},
{
"cell_type": "markdown",
- "id": "ca5a343a",
+ "id": "26c6e4f1",
"metadata": {
"editable": true
},
@@ -1991,7 +2361,7 @@
{
"cell_type": "code",
"execution_count": 10,
- "id": "08d97c1e",
+ "id": "226c13ec",
"metadata": {
"collapsed": false,
"editable": true
@@ -2055,7 +2425,7 @@
},
{
"cell_type": "markdown",
- "id": "727d8fc3",
+ "id": "6895dbfe",
"metadata": {
"editable": true
},
@@ -2070,7 +2440,7 @@
{
"cell_type": "code",
"execution_count": 11,
- "id": "4e41c003",
+ "id": "f8d01982",
"metadata": {
"collapsed": false,
"editable": true
@@ -2154,7 +2524,7 @@
},
{
"cell_type": "markdown",
- "id": "fe00db52",
+ "id": "cffe8367",
"metadata": {
"editable": true
},
@@ -2165,7 +2535,7 @@
{
"cell_type": "code",
"execution_count": 12,
- "id": "8f22105b",
+ "id": "b57871a9",
"metadata": {
"collapsed": false,
"editable": true
@@ -2243,7 +2613,7 @@
},
{
"cell_type": "markdown",
- "id": "8956bf7a",
+ "id": "37b273c1",
"metadata": {
"editable": true
},
@@ -2256,7 +2626,7 @@
{
"cell_type": "code",
"execution_count": 13,
- "id": "044275ef",
+ "id": "9ff0eb69",
"metadata": {
"collapsed": false,
"editable": true
@@ -2300,7 +2670,7 @@
},
{
"cell_type": "markdown",
- "id": "353b50b3",
+ "id": "e7d143b6",
"metadata": {
"editable": true
},
@@ -2311,7 +2681,7 @@
{
"cell_type": "code",
"execution_count": 14,
- "id": "fdc8debd",
+ "id": "9b5e2d1d",
"metadata": {
"collapsed": false,
"editable": true
@@ -2370,7 +2740,7 @@
},
{
"cell_type": "markdown",
- "id": "b738f1b8",
+ "id": "8b6fb13f",
"metadata": {
"editable": true
},
@@ -2380,7 +2750,7 @@
},
{
"cell_type": "markdown",
- "id": "65ce93ba",
+ "id": "06c3f4bb",
"metadata": {
"editable": true
},
@@ -2391,7 +2761,7 @@
{
"cell_type": "code",
"execution_count": 15,
- "id": "604d7286",
+ "id": "4abf9ccd",
"metadata": {
"collapsed": false,
"editable": true
@@ -2456,7 +2826,7 @@
},
{
"cell_type": "markdown",
- "id": "e663a714",
+ "id": "18d42e29",
"metadata": {
"editable": true
},
@@ -2467,7 +2837,7 @@
{
"cell_type": "code",
"execution_count": 16,
- "id": "749fa687",
+ "id": "03415114",
"metadata": {
"collapsed": false,
"editable": true
@@ -2537,7 +2907,7 @@
},
{
"cell_type": "markdown",
- "id": "8801fcd5",
+ "id": "41120d8f",
"metadata": {
"editable": true
},
@@ -2554,7 +2924,7 @@
},
{
"cell_type": "markdown",
- "id": "8ea68725",
+ "id": "16eb2a88",
"metadata": {
"editable": true
},
@@ -2582,7 +2952,7 @@
{
"cell_type": "code",
"execution_count": 17,
- "id": "04811786",
+ "id": "a6df3a5c",
"metadata": {
"collapsed": false,
"editable": true
@@ -2602,7 +2972,7 @@
},
{
"cell_type": "markdown",
- "id": "b0e7cc2c",
+ "id": "fc15d89b",
"metadata": {
"editable": true
},
@@ -2618,7 +2988,7 @@
},
{
"cell_type": "markdown",
- "id": "f8a8132d",
+ "id": "4e4b5ee0",
"metadata": {
"editable": true
},
@@ -2638,7 +3008,7 @@
},
{
"cell_type": "markdown",
- "id": "03eca41f",
+ "id": "4455b9a0",
"metadata": {
"editable": true
},
@@ -2665,7 +3035,7 @@
},
{
"cell_type": "markdown",
- "id": "710e8f88",
+ "id": "9592eb20",
"metadata": {
"editable": true
},
@@ -2678,7 +3048,7 @@
},
{
"cell_type": "markdown",
- "id": "5d3df9bf",
+ "id": "e3022ea8",
"metadata": {
"editable": true
},
@@ -2690,7 +3060,7 @@
},
{
"cell_type": "markdown",
- "id": "be0fd5f1",
+ "id": "246a28bf",
"metadata": {
"editable": true
},
@@ -2705,7 +3075,7 @@
{
"cell_type": "code",
"execution_count": 18,
- "id": "2a0924bb",
+ "id": "d132a060",
"metadata": {
"collapsed": false,
"editable": true
@@ -2732,7 +3102,7 @@
},
{
"cell_type": "markdown",
- "id": "d116f448",
+ "id": "9e70a01f",
"metadata": {
"editable": true
},
@@ -2746,7 +3116,7 @@
},
{
"cell_type": "markdown",
- "id": "41caea07",
+ "id": "2c0ad6b4",
"metadata": {
"editable": true
},
@@ -2758,7 +3128,7 @@
},
{
"cell_type": "markdown",
- "id": "1fa96f7c",
+ "id": "3c6081b8",
"metadata": {
"editable": true
},
@@ -2775,7 +3145,7 @@
},
{
"cell_type": "markdown",
- "id": "70038d6a",
+ "id": "845af933",
"metadata": {
"editable": true
},
@@ -2787,7 +3157,7 @@
},
{
"cell_type": "markdown",
- "id": "852a77d0",
+ "id": "564afbbd",
"metadata": {
"editable": true
},
@@ -2797,7 +3167,7 @@
},
{
"cell_type": "markdown",
- "id": "fc4afaaf",
+ "id": "c4088263",
"metadata": {
"editable": true
},
@@ -2809,7 +3179,7 @@
},
{
"cell_type": "markdown",
- "id": "94b18ced",
+ "id": "96983e3d",
"metadata": {
"editable": true
},
@@ -2819,7 +3189,7 @@
},
{
"cell_type": "markdown",
- "id": "d7a95314",
+ "id": "91d029d7",
"metadata": {
"editable": true
},
@@ -2831,7 +3201,7 @@
},
{
"cell_type": "markdown",
- "id": "eaf6a485",
+ "id": "20d351f6",
"metadata": {
"editable": true
},
@@ -2842,7 +3212,7 @@
},
{
"cell_type": "markdown",
- "id": "3d9442a2",
+ "id": "7a8e79fd",
"metadata": {
"editable": true
},
@@ -2854,7 +3224,7 @@
},
{
"cell_type": "markdown",
- "id": "e4aeef17",
+ "id": "4ec7ad68",
"metadata": {
"editable": true
},
@@ -2864,7 +3234,7 @@
},
{
"cell_type": "markdown",
- "id": "4ce9dee9",
+ "id": "df09c13b",
"metadata": {
"editable": true
},
@@ -2876,7 +3246,7 @@
},
{
"cell_type": "markdown",
- "id": "752ce099",
+ "id": "bb2a9c1f",
"metadata": {
"editable": true
},
@@ -2886,7 +3256,7 @@
},
{
"cell_type": "markdown",
- "id": "7cad5229",
+ "id": "b3507e2d",
"metadata": {
"editable": true
},
@@ -2898,7 +3268,7 @@
},
{
"cell_type": "markdown",
- "id": "46f1aaf9",
+ "id": "2010542f",
"metadata": {
"editable": true
},
@@ -2908,7 +3278,7 @@
},
{
"cell_type": "markdown",
- "id": "7d25a9fb",
+ "id": "e03a1590",
"metadata": {
"editable": true
},
@@ -2920,7 +3290,7 @@
},
{
"cell_type": "markdown",
- "id": "57b4c7d9",
+ "id": "71872755",
"metadata": {
"editable": true
},
@@ -2930,7 +3300,7 @@
},
{
"cell_type": "markdown",
- "id": "fb833214",
+ "id": "167238dc",
"metadata": {
"editable": true
},
@@ -2942,7 +3312,7 @@
},
{
"cell_type": "markdown",
- "id": "5fa29cd3",
+ "id": "38d0cc0f",
"metadata": {
"editable": true
},
@@ -2952,7 +3322,7 @@
},
{
"cell_type": "markdown",
- "id": "6c0e668d",
+ "id": "e9e1beb9",
"metadata": {
"editable": true
},
@@ -2964,7 +3334,7 @@
},
{
"cell_type": "markdown",
- "id": "9d928664",
+ "id": "9a8576e4",
"metadata": {
"editable": true
},
@@ -2974,7 +3344,7 @@
},
{
"cell_type": "markdown",
- "id": "65434b84",
+ "id": "937d703f",
"metadata": {
"editable": true
},
@@ -2986,7 +3356,7 @@
},
{
"cell_type": "markdown",
- "id": "127c9817",
+ "id": "e4723b95",
"metadata": {
"editable": true
},
@@ -2996,7 +3366,7 @@
},
{
"cell_type": "markdown",
- "id": "46f45c10",
+ "id": "6df6f6d8",
"metadata": {
"editable": true
},
@@ -3008,7 +3378,7 @@
},
{
"cell_type": "markdown",
- "id": "4fbaa69a",
+ "id": "39bdaf00",
"metadata": {
"editable": true
},
@@ -3020,7 +3390,7 @@
},
{
"cell_type": "markdown",
- "id": "25f1abd4",
+ "id": "e4584236",
"metadata": {
"editable": true
},
@@ -3032,7 +3402,7 @@
},
{
"cell_type": "markdown",
- "id": "9fd5ef9e",
+ "id": "d0c5d728",
"metadata": {
"editable": true
},
@@ -3042,7 +3412,7 @@
},
{
"cell_type": "markdown",
- "id": "f1cb8e35",
+ "id": "9b637fd2",
"metadata": {
"editable": true
},
@@ -3054,7 +3424,7 @@
},
{
"cell_type": "markdown",
- "id": "c0c5100a",
+ "id": "9627e6fb",
"metadata": {
"editable": true
},
@@ -3067,7 +3437,7 @@
},
{
"cell_type": "markdown",
- "id": "c80e55cb",
+ "id": "662fe97e",
"metadata": {
"editable": true
},
@@ -3079,7 +3449,7 @@
},
{
"cell_type": "markdown",
- "id": "a47f5c5e",
+ "id": "fa2d5cb5",
"metadata": {
"editable": true
},
@@ -3093,7 +3463,7 @@
{
"cell_type": "code",
"execution_count": 19,
- "id": "e093186c",
+ "id": "a530b3ba",
"metadata": {
"collapsed": false,
"editable": true
@@ -3190,7 +3560,7 @@
},
{
"cell_type": "markdown",
- "id": "de555fff",
+ "id": "58cc6d9d",
"metadata": {
"editable": true
},
@@ -3211,7 +3581,7 @@
},
{
"cell_type": "markdown",
- "id": "72178d39",
+ "id": "6a2d3f87",
"metadata": {
"editable": true
},
@@ -3223,7 +3593,7 @@
},
{
"cell_type": "markdown",
- "id": "8e5d822b",
+ "id": "9eab78a9",
"metadata": {
"editable": true
},
@@ -3233,7 +3603,7 @@
},
{
"cell_type": "markdown",
- "id": "e9218f82",
+ "id": "50d7f5c3",
"metadata": {
"editable": true
},
@@ -3245,7 +3615,7 @@
},
{
"cell_type": "markdown",
- "id": "2223d1b1",
+ "id": "4c96e589",
"metadata": {
"editable": true
},
@@ -3255,7 +3625,7 @@
},
{
"cell_type": "markdown",
- "id": "e5474a5b",
+ "id": "b57830ea",
"metadata": {
"editable": true
},
@@ -3267,7 +3637,7 @@
},
{
"cell_type": "markdown",
- "id": "691295ed",
+ "id": "05325bc6",
"metadata": {
"editable": true
},
@@ -3285,7 +3655,7 @@
{
"cell_type": "code",
"execution_count": 20,
- "id": "e243cef5",
+ "id": "d95b15bc",
"metadata": {
"collapsed": false,
"editable": true
@@ -3361,7 +3731,7 @@
},
{
"cell_type": "markdown",
- "id": "ef2eaa7a",
+ "id": "b88ebede",
"metadata": {
"editable": true
},
@@ -3375,7 +3745,7 @@
{
"cell_type": "code",
"execution_count": 21,
- "id": "546e3504",
+ "id": "47036b16",
"metadata": {
"collapsed": false,
"editable": true
@@ -3464,7 +3834,7 @@
},
{
"cell_type": "markdown",
- "id": "f6787352",
+ "id": "52faee2f",
"metadata": {
"editable": true
},
diff --git a/doc/src/week37/Latexfiles/sgd.txt b/doc/src/week37/Latexfiles/sgd.txt
new file mode 100644
index 000000000..3502d6192
--- /dev/null
+++ b/doc/src/week37/Latexfiles/sgd.txt
@@ -0,0 +1,246 @@
+!split
+===== SGD vs Full-Batch GD: Convergence Speed and Memory Comparison =====
+
+
+
+=== Theoretical Convergence Speed and convex optimization ===
+
+
+Consider minimizing an empirical cost function
+!bt
+\[
+C(\theta) =\frac{1}{N}\sum_{i=1}^N l_i(\theta),
+\]
+!et
+
+where each $l_i(\theta)$ is a
+differentiable loss term. Gradient Descent (GD) updates parameters
+using the full gradient $\nabla C(\theta)$, while Stochastic Gradient
+Descent (SGD) uses a single sample (or mini-batch) gradient $\nabla
+l_i(\theta)$ selected at random. In equation form, one GD step is:
+
+!bt
+\[
+\theta_{t+1} = \theta_t-\eta \nabla C(\theta_t) =\theta_t -\eta \frac{1}{N}\sum_{i=1}^N \nabla l_i(\theta_t),
+\]
+!et
+whereas one SGD step is:
+
+!bt
+\[
+\theta_{t+1} ;=; \theta_t -\eta \nabla l_{i_t}(\theta_t),
+\]
+!et
+
+with $i_t$ randomly chosen. On smooth convex problems, GD and SGD both
+converge to the global minimum, but their rates differ. GD can take
+larger, more stable steps since it uses the exact gradient, achieving
+an error that decreases on the order of $O(1/t)$ per iteration for
+convex objectives (and even exponentially fast for strongly convex
+cases). In contrast, plain SGD has more variance in each step, leading
+to sublinear convergence in expectation – typically $O(1/\sqrt{t})$
+for general convex objectives (\thetaith appropriate diminishing step
+sizes) . Intuitively, GD’s trajectory is smoother and more
+predictable, while SGD’s path oscillates due to noise but costs far
+less per iteration, enabling many more updates in the same time.
+
+
+=== Strongly Convex Case ===
+
+
+If $C(\theta)$ is strongly convex and $L$-smooth (so GD enjoys linear
+convergence), the gap $C(\theta_t)-C(\theta^*)$ for GD shrinks as
+!bt
+\[
+C(\theta_t) - C(\theta^* ) ;\le; \Big(1 - \frac{\mu}{L}\Big)^t [C(\theta_0)-C(\theta^*)],
+\]
+!et
+
+a geometric (linear) convergence per iteration . Achieving an
+$\epsilon$-accurate solution thus takes on the order of
+$\log(1/\epsilon)$ iterations for GD. However, each GD iteration costs
+$O(N)$ gradient evaluations. SGD cannot exploit strong convexity to
+obtain a linear rate – instead, with a properly decaying step size
+(e.g. $\eta_t = \frac{1}{\mu t}$) or iterate averaging, SGD attains an
+$O(1/t)$ convergence rate in expectation . For example, one result
+of Moulines and Bach 2011, see URL:"https://papers.nips.cc/paper_files/paper/2011/hash/40008b9a5380fcacce3976bf7c08af5b-Abstract.html" shows that with $\eta_t = \Theta(1/t)$,
+!bt
+\[
+\mathbb{E}[C(\theta_t) - C(\theta^*)] = O(1/t),
+\]
+!et
+
+for strongly convex, smooth $F$ . This $1/t$ rate is slower per
+iteration than GD’s exponential decay, but each SGD iteration is $N$
+times cheaper. In fact, to reach error $\epsilon$, plain SGD needs on
+the order of $T=O(1/\epsilon)$ iterations (sub-linear convergence),
+while GD needs $O(\log(1/\epsilon))$ iterations. When accounting for
+cost-per-iteration, GD requires $O(N \log(1/\epsilon))$ total gradient
+computations versus SGD’s $O(1/\epsilon)$ single-sample
+computations. In large-scale regimes (huge $N$), SGD can be
+faster in wall-clock time because $N \log(1/\epsilon)$ may far exceed
+$1/\epsilon$ for reasonable accuracy levels. In other words,
+with millions of data points, one epoch of GD (one full gradient) is
+extremely costly, whereas SGD can make $N$ cheap updates in the time
+GD makes one – often yielding a good solution faster in practice, even
+though SGD’s asymptotic error decays more slowly. As one lecture
+succinctly puts it: “SGD can be super effective in terms of iteration
+cost and memory, but SGD is slow to converge and can’t adapt to strong
+convexity” . Thus, the break-even point depends on $N$ and the desired
+accuracy: for moderate accuracy on very large $N$, SGD’s cheaper
+updates win; for extremely high precision (very small $\epsilon$) on a
+modest $N$, GD’s fast convergence per step can be advantageous.
+
+=== Non-Convex Problems ===
+
+In non-convex optimization (e.g. deep neural networks), neither GD nor
+SGD guarantees global minima, but SGD often displays faster progress
+in finding useful minima. Theoretical results here are weaker, usually
+showing convergence to a stationary point (\thetahere $|\nabla F|$ is
+small) in expectation. For example, GD might require $O(1/\epsilon^2)$
+iterations to ensure $|\nabla C(\theta)| < \epsilon$, and SGD typically has
+similar polynomial complexity (often worse due to gradient
+noise). However, a noteworthy difference is that SGD’s stochasticity
+can help escape saddle points or poor local minima. Random gradient
+fluctuations act like implicit noise, helping the iterate “jump” out
+of flat saddle regions where full-batch GD could stagnate . In fact,
+research has shown that adding noise to GD can guarantee escaping
+saddle points in polynomial time, and the inherent noise in SGD often
+serves this role. Empirically, this means SGD can sometimes find a
+lower loss basin faster, whereas full-batch GD might get “stuck” near
+saddle points or need a very small learning rate to navigate complex
+error surfaces . Overall, in modern high-dimensional machine learning,
+SGD (or mini-batch SGD) is the workhorse for large non-convex problems
+because it converges to good solutions much faster in practice,
+despite the lack of a linear convergence guarantee. Full-batch GD is
+rarely used on large neural networks, as it would require tiny steps
+to avoid divergence and is extremely slow per iteration .
+
+!split
+===== Memory Usage and Scalability =====
+
+
+A major advantage of SGD is its memory efficiency in handling large
+datasets. Full-batch GD requires access to the entire training set for
+each iteration, which often means the whole dataset (or a large
+subset) must reside in memory to compute $\nabla C(\theta)$ . This results
+in memory usage that scales linearly with the dataset size $N$. For
+instance, if each training sample is large (e.g. high-dimensional
+features), computing a full gradient may require storing a substantial
+portion of the data or all intermediate gradients until they are
+aggregated. In contrast, SGD needs only a single (or a small
+mini-batch of) training example(s) in memory at any time . The
+algorithm processes one sample (or mini-batch) at a time and
+immediately updates the model, discarding that sample before moving to
+the next. This streaming approach means that memory footprint is
+essentially independent of $N$ (apart from storing the model
+parameters themselves). As one source notes, gradient descent
+“requires more memory than SGD” because it “must store the entire
+dataset for each iteration,” whereas SGD “only needs to store the
+current training example” . In practical terms, if you have a dataset
+of size, say, 1 million examples, full-batch GD would need memory for
+all million every step, while SGD could be implemented to load just
+one example at a time – a crucial benefit if data are too large to fit
+in RAM or GPU memory. This scalability makes SGD suitable for
+large-scale learning: as long as you can stream data from disk, SGD
+can handle arbitrarily large datasets with fixed memory. In fact, SGD
+“does not need to remember which examples were visited” in the past,
+allowing it to run in an online fashion on infinite data streams
+. Full-batch GD, on the other hand, would require multiple passes
+through a giant dataset per update (or a complex distributed memory
+system), which is often infeasible.
+
+There is also a secondary memory effect: computing a full-batch
+gradient in deep learning requires storing all intermediate
+activations for backpropagation across the entire batch. A very large
+batch (approaching the full dataset) might exhaust GPU memory due to
+the need to hold activation gradients for thousands or millions of
+examples simultaneously. SGD/minibatches mitigate this by splitting
+the workload – e.g. with a mini-batch of size 32 or 256, memory use
+stays bounded, whereas a full-batch (size = $N$) forward/backward pass
+could not even be executed if $N$ is huge. Techniques like gradient
+accumulation exist to simulate large-batch GD by summing many
+small-batch gradients – but these still process data in manageable
+chunks to avoid memory overflow. In summary, memory complexity for GD
+grows with $N$, while for SGD it remains $O(1)$ w.r.t. dataset size
+(only the model and perhaps a mini-batch reside in memory) . This is a
+key reason why batch GD “does not scale” to very large data and why
+virtually all large-scale machine learning algorithms rely on
+stochastic or mini-batch methods.
+
+
+!split
+===== Empirical Evidence: Convergence Time and Memory in Practice =====
+
+
+Empirical studies strongly support the theoretical trade-offs
+above. In large-scale machine learning tasks, SGD often converges to a
+good solution much faster in wall-clock time than full-batch GD, and
+it uses far less memory. For example, Bottou & Bousquet (2008)
+analyzed learning time under a fixed computational budget and
+concluded that when data is abundant, it’s better to use a faster
+(even if less precise) optimization method to process more examples in
+the same time . This analysis showed that for large-scale problems,
+processing more data with SGD yields lower error than spending the
+time to do exact (batch) optimization on fewer data . In other words,
+if you have a time budget, it’s often optimal to accept slightly
+slower convergence per step (as with SGD) in exchange for being able
+to use many more training samples in that time. This phenomenon is
+borne out by experiments:
+
+
+
+=== Deep Neural Networks ===
+
+In modern deep learning, full-batch GD is so slow that it is rarely
+attempted; instead, mini-batch SGD is standard. A recent study
+demonstrated that it is possible to train a ResNet-50 on ImageNet
+using full-batch gradient descent, but it required careful tuning
+(e.g. gradient clipping, tiny learning rates) and vast computational
+resources – and even then, each full-batch update was extremely
+expensive.
+
+Using a huge batch
+(closer to full GD) tends to slow down convergence if the learning
+rate is not scaled up, and often encounters optimization difficulties
+(plateaus) that small batches avoid.
+Empirically, small or medium
+batch SGD finds minima in fewer clock hours because it can rapidly
+loop over the data with gradient noise aiding exploration.
+
+=== Memory constraints ===
+
+From a memory standpoint, practitioners note that batch GD becomes
+infeasible on large data. For example, if one tried to do full-batch
+training on a dataset that doesn’t fit in RAM or GPU memory, the
+program would resort to heavy disk I/O or simply crash. SGD
+circumvents this by processing mini-batches. Even in cases where data
+does fit in memory, using a full batch can spike memory usage due to
+storing all gradients. One empirical observation is that mini-batch
+training has a “lower, fluctuating usage pattern” of memory, whereas
+full-batch loading “quickly consumes memory (often exceeding limits)”
+. This is especially relevant for graph neural networks or other
+models where a “batch” may include a huge chunk of a graph: full-batch
+gradient computation can exhaust GPU memory, whereas mini-batch
+methods keep memory usage manageable .
+
+
+In summary, SGD converges faster than full-batch GD in terms of actual
+training time for large-scale problems, provided we measure
+convergence as reaching a good-enough solution. Theoretical bounds
+show SGD needs more iterations, but because it performs many more
+updates per unit time (and requires far less memory), it often
+achieves lower loss in a given time frame than GD. Full-batch GD might
+take slightly fewer iterations in theory, but each iteration is so
+costly that it is “slower… especially for large datasets” . Meanwhile,
+memory scaling strongly favors SGD: GD’s memory cost grows with
+dataset size, making it impractical beyond a point, whereas SGD’s
+memory use is modest and mostly constant w.r.t. $N$ . These
+differences have made SGD (and mini-batch variants) the de facto
+choice for training large machine learning models, from logistic
+regression on millions of examples to deep neural networks with
+billions of parameters. The consensus in both research and practice is
+that for large-scale or high-dimensional tasks, SGD-type methods
+converge quicker per unit of computation and handle memory constraints
+better than standard full-batch gradient descent .
+
diff --git a/doc/src/week37/week37.do.txt b/doc/src/week37/week37.do.txt
index f2abeb09c..4b2a9c628 100644
--- a/doc/src/week37/week37.do.txt
+++ b/doc/src/week37/week37.do.txt
@@ -794,6 +794,252 @@ mini-batches. The discussion
"here":"https://sebastianraschka.com/faq/docs/sgd-methods.html" may be
useful.
+!split
+===== SGD vs Full-Batch GD: Convergence Speed and Memory Comparison =====
+
+
+
+=== Theoretical Convergence Speed and convex optimization ===
+
+
+Consider minimizing an empirical cost function
+!bt
+\[
+C(\theta) =\frac{1}{N}\sum_{i=1}^N l_i(\theta),
+\]
+!et
+
+where each $l_i(\theta)$ is a
+differentiable loss term. Gradient Descent (GD) updates parameters
+using the full gradient $\nabla C(\theta)$, while Stochastic Gradient
+Descent (SGD) uses a single sample (or mini-batch) gradient $\nabla
+l_i(\theta)$ selected at random. In equation form, one GD step is:
+
+!bt
+\[
+\theta_{t+1} = \theta_t-\eta \nabla C(\theta_t) =\theta_t -\eta \frac{1}{N}\sum_{i=1}^N \nabla l_i(\theta_t),
+\]
+!et
+whereas one SGD step is:
+
+!bt
+\[
+\theta_{t+1} = \theta_t -\eta \nabla l_{i_t}(\theta_t),
+\]
+!et
+
+with $i_t$ randomly chosen. On smooth convex problems, GD and SGD both
+converge to the global minimum, but their rates differ. GD can take
+larger, more stable steps since it uses the exact gradient, achieving
+an error that decreases on the order of $O(1/t)$ per iteration for
+convex objectives (and even exponentially fast for strongly convex
+cases). In contrast, plain SGD has more variance in each step, leading
+to sublinear convergence in expectation – typically $O(1/\sqrt{t})$
+for general convex objectives (\thetaith appropriate diminishing step
+sizes) . Intuitively, GD’s trajectory is smoother and more
+predictable, while SGD’s path oscillates due to noise but costs far
+less per iteration, enabling many more updates in the same time.
+
+
+=== Strongly Convex Case ===
+
+
+If $C(\theta)$ is strongly convex and $L$-smooth (so GD enjoys linear
+convergence), the gap $C(\theta_t)-C(\theta^*)$ for GD shrinks as
+!bt
+\[
+C(\theta_t) - C(\theta^* ) \le \Big(1 - \frac{\mu}{L}\Big)^t [C(\theta_0)-C(\theta^*)],
+\]
+!et
+
+a geometric (linear) convergence per iteration . Achieving an
+$\epsilon$-accurate solution thus takes on the order of
+$\log(1/\epsilon)$ iterations for GD. However, each GD iteration costs
+$O(N)$ gradient evaluations. SGD cannot exploit strong convexity to
+obtain a linear rate – instead, with a properly decaying step size
+(e.g. $\eta_t = \frac{1}{\mu t}$) or iterate averaging, SGD attains an
+$O(1/t)$ convergence rate in expectation . For example, one result
+of Moulines and Bach 2011, see URL:"https://papers.nips.cc/paper_files/paper/2011/hash/40008b9a5380fcacce3976bf7c08af5b-Abstract.html" shows that with $\eta_t = \Theta(1/t)$,
+!bt
+\[
+\mathbb{E}[C(\theta_t) - C(\theta^*)] = O(1/t),
+\]
+!et
+
+for strongly convex, smooth $F$ . This $1/t$ rate is slower per
+iteration than GD’s exponential decay, but each SGD iteration is $N$
+times cheaper. In fact, to reach error $\epsilon$, plain SGD needs on
+the order of $T=O(1/\epsilon)$ iterations (sub-linear convergence),
+while GD needs $O(\log(1/\epsilon))$ iterations. When accounting for
+cost-per-iteration, GD requires $O(N \log(1/\epsilon))$ total gradient
+computations versus SGD’s $O(1/\epsilon)$ single-sample
+computations. In large-scale regimes (huge $N$), SGD can be
+faster in wall-clock time because $N \log(1/\epsilon)$ may far exceed
+$1/\epsilon$ for reasonable accuracy levels. In other words,
+with millions of data points, one epoch of GD (one full gradient) is
+extremely costly, whereas SGD can make $N$ cheap updates in the time
+GD makes one – often yielding a good solution faster in practice, even
+though SGD’s asymptotic error decays more slowly. As one lecture
+succinctly puts it: “SGD can be super effective in terms of iteration
+cost and memory, but SGD is slow to converge and can’t adapt to strong
+convexity” . Thus, the break-even point depends on $N$ and the desired
+accuracy: for moderate accuracy on very large $N$, SGD’s cheaper
+updates win; for extremely high precision (very small $\epsilon$) on a
+modest $N$, GD’s fast convergence per step can be advantageous.
+
+=== Non-Convex Problems ===
+
+In non-convex optimization (e.g. deep neural networks), neither GD nor
+SGD guarantees global minima, but SGD often displays faster progress
+in finding useful minima. Theoretical results here are weaker, usually
+showing convergence to a stationary point $\theta$ ($|\nabla C|$ is
+small) in expectation. For example, GD might require $O(1/\epsilon^2)$
+iterations to ensure $|\nabla C(\theta)| < \epsilon$, and SGD typically has
+similar polynomial complexity (often worse due to gradient
+noise). However, a noteworthy difference is that SGD’s stochasticity
+can help escape saddle points or poor local minima. Random gradient
+fluctuations act like implicit noise, helping the iterate “jump” out
+of flat saddle regions where full-batch GD could stagnate . In fact,
+research has shown that adding noise to GD can guarantee escaping
+saddle points in polynomial time, and the inherent noise in SGD often
+serves this role. Empirically, this means SGD can sometimes find a
+lower loss basin faster, whereas full-batch GD might get “stuck” near
+saddle points or need a very small learning rate to navigate complex
+error surfaces . Overall, in modern high-dimensional machine learning,
+SGD (or mini-batch SGD) is the workhorse for large non-convex problems
+because it converges to good solutions much faster in practice,
+despite the lack of a linear convergence guarantee. Full-batch GD is
+rarely used on large neural networks, as it would require tiny steps
+to avoid divergence and is extremely slow per iteration .
+
+!split
+===== Memory Usage and Scalability =====
+
+
+A major advantage of SGD is its memory efficiency in handling large
+datasets. Full-batch GD requires access to the entire training set for
+each iteration, which often means the whole dataset (or a large
+subset) must reside in memory to compute $\nabla C(\theta)$ . This results
+in memory usage that scales linearly with the dataset size $N$. For
+instance, if each training sample is large (e.g. high-dimensional
+features), computing a full gradient may require storing a substantial
+portion of the data or all intermediate gradients until they are
+aggregated. In contrast, SGD needs only a single (or a small
+mini-batch of) training example(s) in memory at any time . The
+algorithm processes one sample (or mini-batch) at a time and
+immediately updates the model, discarding that sample before moving to
+the next. This streaming approach means that memory footprint is
+essentially independent of $N$ (apart from storing the model
+parameters themselves). As one source notes, gradient descent
+“requires more memory than SGD” because it “must store the entire
+dataset for each iteration,” whereas SGD “only needs to store the
+current training example” . In practical terms, if you have a dataset
+of size, say, 1 million examples, full-batch GD would need memory for
+all million every step, while SGD could be implemented to load just
+one example at a time – a crucial benefit if data are too large to fit
+in RAM or GPU memory. This scalability makes SGD suitable for
+large-scale learning: as long as you can stream data from disk, SGD
+can handle arbitrarily large datasets with fixed memory. In fact, SGD
+“does not need to remember which examples were visited” in the past,
+allowing it to run in an online fashion on infinite data streams
+. Full-batch GD, on the other hand, would require multiple passes
+through a giant dataset per update (or a complex distributed memory
+system), which is often infeasible.
+
+There is also a secondary memory effect: computing a full-batch
+gradient in deep learning requires storing all intermediate
+activations for backpropagation across the entire batch. A very large
+batch (approaching the full dataset) might exhaust GPU memory due to
+the need to hold activation gradients for thousands or millions of
+examples simultaneously. SGD/minibatches mitigate this by splitting
+the workload – e.g. with a mini-batch of size 32 or 256, memory use
+stays bounded, whereas a full-batch (size = $N$) forward/backward pass
+could not even be executed if $N$ is huge. Techniques like gradient
+accumulation exist to simulate large-batch GD by summing many
+small-batch gradients – but these still process data in manageable
+chunks to avoid memory overflow. In summary, memory complexity for GD
+grows with $N$, while for SGD it remains $O(1)$ w.r.t. dataset size
+(only the model and perhaps a mini-batch reside in memory) . This is a
+key reason why batch GD “does not scale” to very large data and why
+virtually all large-scale machine learning algorithms rely on
+stochastic or mini-batch methods.
+
+
+!split
+===== Empirical Evidence: Convergence Time and Memory in Practice =====
+
+
+Empirical studies strongly support the theoretical trade-offs
+above. In large-scale machine learning tasks, SGD often converges to a
+good solution much faster in wall-clock time than full-batch GD, and
+it uses far less memory. For example, Bottou & Bousquet (2008)
+analyzed learning time under a fixed computational budget and
+concluded that when data is abundant, it’s better to use a faster
+(even if less precise) optimization method to process more examples in
+the same time . This analysis showed that for large-scale problems,
+processing more data with SGD yields lower error than spending the
+time to do exact (batch) optimization on fewer data . In other words,
+if you have a time budget, it’s often optimal to accept slightly
+slower convergence per step (as with SGD) in exchange for being able
+to use many more training samples in that time. This phenomenon is
+borne out by experiments:
+
+
+
+=== Deep Neural Networks ===
+
+In modern deep learning, full-batch GD is so slow that it is rarely
+attempted; instead, mini-batch SGD is standard. A recent study
+demonstrated that it is possible to train a ResNet-50 on ImageNet
+using full-batch gradient descent, but it required careful tuning
+(e.g. gradient clipping, tiny learning rates) and vast computational
+resources – and even then, each full-batch update was extremely
+expensive.
+
+Using a huge batch
+(closer to full GD) tends to slow down convergence if the learning
+rate is not scaled up, and often encounters optimization difficulties
+(plateaus) that small batches avoid.
+Empirically, small or medium
+batch SGD finds minima in fewer clock hours because it can rapidly
+loop over the data with gradient noise aiding exploration.
+
+=== Memory constraints ===
+
+From a memory standpoint, practitioners note that batch GD becomes
+infeasible on large data. For example, if one tried to do full-batch
+training on a dataset that doesn’t fit in RAM or GPU memory, the
+program would resort to heavy disk I/O or simply crash. SGD
+circumvents this by processing mini-batches. Even in cases where data
+does fit in memory, using a full batch can spike memory usage due to
+storing all gradients. One empirical observation is that mini-batch
+training has a “lower, fluctuating usage pattern” of memory, whereas
+full-batch loading “quickly consumes memory (often exceeding limits)”
+. This is especially relevant for graph neural networks or other
+models where a “batch” may include a huge chunk of a graph: full-batch
+gradient computation can exhaust GPU memory, whereas mini-batch
+methods keep memory usage manageable .
+
+
+In summary, SGD converges faster than full-batch GD in terms of actual
+training time for large-scale problems, provided we measure
+convergence as reaching a good-enough solution. Theoretical bounds
+show SGD needs more iterations, but because it performs many more
+updates per unit time (and requires far less memory), it often
+achieves lower loss in a given time frame than GD. Full-batch GD might
+take slightly fewer iterations in theory, but each iteration is so
+costly that it is “slower… especially for large datasets” . Meanwhile,
+memory scaling strongly favors SGD: GD’s memory cost grows with
+dataset size, making it impractical beyond a point, whereas SGD’s
+memory use is modest and mostly constant w.r.t. $N$ . These
+differences have made SGD (and mini-batch variants) the de facto
+choice for training large machine learning models, from logistic
+regression on millions of examples to deep neural networks with
+billions of parameters. The consensus in both research and practice is
+that for large-scale or high-dimensional tasks, SGD-type methods
+converge quicker per unit of computation and handle memory constraints
+better than standard full-batch gradient descent .
+
!split