diff --git a/doc/pub/week39/html/week39-bs.html b/doc/pub/week39/html/week39-bs.html index 27c968c67..962307c78 100644 --- a/doc/pub/week39/html/week39-bs.html +++ b/doc/pub/week39/html/week39-bs.html @@ -178,46 +178,10 @@ doconce format html week39.do.txt --html_style=bootstrap --pygments_html_style=d 2, None, 'program-for-stochastic-gradient'), - ('Momentum based GD', 2, None, 'momentum-based-gd'), - ('More on momentum based approaches', + ('Code with a Number of Minibatches which varies', 2, None, - 'more-on-momentum-based-approaches'), - ('Momentum parameter', 2, None, 'momentum-parameter'), - ('Second moment of the gradient', - 2, - None, - 'second-moment-of-the-gradient'), - ('RMS prop', 2, None, 'rms-prop'), - ('ADAM optimizer', 2, None, 'adam-optimizer'), - ('Practical tips', 2, None, 'practical-tips'), - ('Automatic differentiation', - 2, - None, - 'automatic-differentiation'), - ('Using autograd', 2, None, 'using-autograd'), - ('Autograd with more complicated functions', - 2, - None, - 'autograd-with-more-complicated-functions'), - ('More complicated functions using the elements of their ' - 'arguments directly', - 2, - None, - 'more-complicated-functions-using-the-elements-of-their-arguments-directly'), - ('Functions using mathematical functions from Numpy', - 2, - None, - 'functions-using-mathematical-functions-from-numpy'), - ('More autograd', 2, None, 'more-autograd'), - ('And with loops', 2, None, 'and-with-loops'), - ('Using recursion', 2, None, 'using-recursion'), - ('Unsupported functions', 2, None, 'unsupported-functions'), - ('The syntax a.dot(b) when finding the dot product', - 2, - None, - 'the-syntax-a-dot-b-when-finding-the-dot-product'), - ('Recommended to avoid', 2, None, 'recommended-to-avoid')]} + 'code-with-a-number-of-minibatches-which-varies')]} end of tocinfo -->
@@ -309,24 +273,7 @@ MathJax.Hub.Config({The stochastic gradient descent (SGD) is almost always used with a -momentum or inertia term that serves as a memory of the direction we -are moving in parameter space. This is typically implemented as -follows -
+In the code here we vary the number of mini-batches.
-
-$$
-\begin{align}
-\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t) \nonumber \\
-\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t},
-\tag{2}
-\end{align}
-$$
-
-
-
where we have introduced a momentum parameter \( \gamma \), with -\( 0\le\gamma\le 1 \), and for brevity we dropped the explicit notation to -indicate the gradient is to be taken over a different mini-batch at -each step. We call this algorithm gradient descent with momentum -(GDM). From these equations, it is clear that \( \mathbf{v}_t \) is a -running average of recently encountered gradients and -\( (1-\gamma)^{-1} \) sets the characteristic time scale for the memory -used in the averaging procedure. Consistent with this, when -\( \gamma=0 \), this just reduces down to ordinary SGD as discussed -earlier. An equivalent way of writing the updates is -
- -
-$$
-\Delta \boldsymbol{\theta}_{t+1} = \gamma \Delta \boldsymbol{\theta}_t -\ \eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t),
-$$
-
-
-
where we have defined \( \Delta \boldsymbol{\theta}_{t}= \boldsymbol{\theta}_t-\boldsymbol{\theta}_{t-1} \).
-Let us try to get more intuition from these equations. It is helpful -to consider a simple physical analogy with a particle of mass \( m \) -moving in a viscous medium with drag coefficient \( \mu \) and potential -\( E(\mathbf{w}) \). If we denote the particle's position by \( \mathbf{w} \), -then its motion is described by -
- -
-$$
-m {d^2 \mathbf{w} \over dt^2} + \mu {d \mathbf{w} \over dt }= -\nabla_w E(\mathbf{w}).
-$$
-
-
-
We can discretize this equation in the usual way to get
- -
-$$
-m { \mathbf{w}_{t+\Delta t}-2 \mathbf{w}_{t} +\mathbf{w}_{t-\Delta t} \over (\Delta t)^2}+\mu {\mathbf{w}_{t+\Delta t}- \mathbf{w}_{t} \over \Delta t} = -\nabla_w E(\mathbf{w}).
-$$
-
-
-
Rearranging this equation, we can rewrite this as
- -
-$$
-\Delta \mathbf{w}_{t +\Delta t}= - { (\Delta t)^2 \over m +\mu \Delta t} \nabla_w E(\mathbf{w})+ {m \over m +\mu \Delta t} \Delta \mathbf{w}_t.
-$$
-
-
Notice that this equation is identical to previous one if we identify -the position of the particle, \( \mathbf{w} \), with the parameters -\( \boldsymbol{\theta} \). This allows us to identify the momentum -parameter and learning rate with the mass of the particle and the -viscous drag as: -
- -
-$$
-\gamma= {m \over m +\mu \Delta t }, \qquad \eta = {(\Delta t)^2 \over m +\mu \Delta t}.
-$$
-
-
-
Thus, as the name suggests, the momentum parameter is proportional to -the mass of the particle and effectively provides inertia. -Furthermore, in the large viscosity/small learning rate limit, our -memory time scales as \( (1-\gamma)^{-1} \approx m/(\mu \Delta t) \). -
- -Why is momentum useful? SGD momentum helps the gradient descent -algorithm gain speed in directions with persistent but small gradients -even in the presence of stochasticity, while suppressing oscillations -in high-curvature directions. This becomes especially important in -situations where the landscape is shallow and flat in some directions -and narrow and steep in others. It has been argued that first-order -methods (with appropriate initial conditions) can perform comparable -to more expensive second order methods, especially in the context of -complex deep learning models. -
- -These beneficial properties of momentum can sometimes become even more -pronounced by using a slight modification of the classical momentum -algorithm called Nesterov Accelerated Gradient (NAG). -
- -In the NAG algorithm, rather than calculating the gradient at the -current parameters, \( \nabla_\theta E(\boldsymbol{\theta}_t) \), one -calculates the gradient at the expected value of the parameters given -our current momentum, \( \nabla_\theta E(\boldsymbol{\theta}_t +\gamma -\mathbf{v}_{t-1}) \). This yields the NAG update rule -
- -
-$$
-\begin{align}
-\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t +\gamma \mathbf{v}_{t-1}) \nonumber \\
-\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t}.
-\tag{3}
-\end{align}
-$$
-
-
-
One of the major advantages of NAG is that it allows for the use of a larger learning rate than GDM for the same choice of \( \gamma \).
-In stochastic gradient descent, with and without momentum, we still -have to specify a schedule for tuning the learning rates \( \eta_t \) -as a function of time. As discussed in the context of Newton's -method, this presents a number of dilemmas. The learning rate is -limited by the steepest direction which can change depending on the -current position in the landscape. To circumvent this problem, ideally -our algorithm would keep track of curvature and take large steps in -shallow, flat directions and small steps in steep, narrow directions. -Second-order methods accomplish this by calculating or approximating -the Hessian and normalizing the learning rate by the -curvature. However, this is very computationally expensive for -extremely large models. Ideally, we would like to be able to -adaptively change the step size to match the landscape without paying -the steep computational price of calculating or approximating -Hessians. -
- -Recently, a number of methods have been introduced that accomplish -this by tracking not only the gradient, but also the second moment of -the gradient. These methods include AdaGrad, AdaDelta, RMS-Prop, and -ADAM. -
-In RMS prop, in addition to keeping a running average of the first -moment of the gradient, we also keep track of the second moment -denoted by \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}_t^2] \). The update rule -for RMS prop is given by -
- -
-$$
-\begin{align}
-\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta})
-\tag{4}\\
-\mathbf{s}_t &=\beta \mathbf{s}_{t-1} +(1-\beta)\mathbf{g}_t^2 \nonumber \\
-\boldsymbol{\theta}_{t+1}&=&\boldsymbol{\theta}_t - \eta_t { \mathbf{g}_t \over \sqrt{\mathbf{s}_t +\epsilon}}, \nonumber
-\end{align}
-$$
-
-
-
where \( \beta \) controls the averaging time of the second moment and is -typically taken to be about \( \beta=0.9 \), \( \eta_t \) is a learning rate -typically chosen to be \( 10^{-3} \), and \( \epsilon\sim 10^{-8} \) is a -small regularization constant to prevent divergences. Multiplication -and division by vectors is understood as an element-wise operation. It -is clear from this formula that the learning rate is reduced in -directions where the norm of the gradient is consistently large. This -greatly speeds up the convergence by allowing us to use a larger -learning rate for flat directions. -
-A related algorithm is the ADAM optimizer. In ADAM, we keep a running -average of both the first and second moment of the gradient and use -this information to adaptively change the learning rate for different -parameters. In addition to keeping a running average of the first and -second moments of the gradient -(i.e. \( \mathbf{m}_t=\mathbb{E}[\mathbf{g}_t] \) and -\( \mathbf{s}_t=\mathbb{E}[\mathbf{g}^2_t] \), respectively), ADAM -performs an additional bias correction to account for the fact that we -are estimating the first two moments of the gradient using a running -average (denoted by the hats in the update rule below). The update -rule for ADAM is given by (where multiplication and division are once -again understood to be element-wise operations below) -
- -
-$$
-\begin{align}
-\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta})
-\tag{5}\\
-\mathbf{m}_t &= \beta_1 \mathbf{m}_{t-1} + (1-\beta_1) \mathbf{g}_t \nonumber \\
-\mathbf{s}_t &=\beta_2 \mathbf{s}_{t-1} +(1-\beta_2)\mathbf{g}_t^2 \nonumber \\
-\boldsymbol{\mathbf{m}}_t&={\mathbf{m}_t \over 1-\beta_1^t} \nonumber \\
-\boldsymbol{\mathbf{s}}_t &={\mathbf{s}_t \over1-\beta_2^t} \nonumber \\
-\boldsymbol{\theta}_{t+1}&=\boldsymbol{\theta}_t - \eta_t { \boldsymbol{\mathbf{m}}_t \over \sqrt{\boldsymbol{\mathbf{s}}_t} +\epsilon}, \nonumber \\
-\tag{6}
-\end{align}
-$$
-
-
-
where \( \beta_1 \) and \( \beta_2 \) set the memory lifetime of the first and -second moment and are typically taken to be \( 0.9 \) and \( 0.99 \) -respectively, and \( \eta \) and \( \epsilon \) are identical to RMSprop. -
- -Like in RMSprop, the effective step size of a parameter depends on the -magnitude of its gradient squared. To understand this better, let us -rewrite this expression in terms of the variance -\( \boldsymbol{\sigma}_t^2 = \boldsymbol{\mathbf{s}}_t - -(\boldsymbol{\mathbf{m}}_t)^2 \). Consider a single parameter \( \theta_t \). The -update rule for this parameter is given by -
- -
-$$
-\Delta \theta_{t+1}= -\eta_t { \boldsymbol{m}_t \over \sqrt{\sigma_t^2 + m_t^2 }+\epsilon}.
-$$
-
-
-
Geron's text, see chapter 11, has several interesting discussions.
-Automatic differentiation (AD), -also called algorithmic -differentiation or computational differentiation,is a set of -techniques to numerically evaluate the derivative of a function -specified by a computer program. AD exploits the fact that every -computer program, no matter how complicated, executes a sequence of -elementary arithmetic operations (addition, subtraction, -multiplication, division, etc.) and elementary functions (exp, log, -sin, cos, etc.). By applying the chain rule repeatedly to these -operations, derivatives of arbitrary order can be computed -automatically, accurately to working precision, and using at most a -small constant factor more arithmetic operations than the original -program. -
- -Automatic differentiation is neither:
- --
Symbolic differentiation can lead to inefficient code and faces the -difficulty of converting a computer program into a single expression, -while numerical differentiation can introduce round-off errors in the -discretization process and cancellation -
- -Python has tools for so-called automatic differentiation. -Consider the following example -
-
-$$
-f(x) = \sin\left(2\pi x + x^2\right)
-$$
-
-
-
which has the following derivative
-
-$$
-f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right)
-$$
-
-
-
Using autograd we have
- - - +import autograd.numpy as np
+ # Importing various packages
+from math import exp, sqrt
+from random import random, seed
+import numpy as np
+import matplotlib.pyplot as plt
-# To do elementwise differentiation:
-from autograd import elementwise_grad as egrad
+n = 100
+x = 2*np.random.rand(n,1)
+y = 4+3*x+np.random.randn(n,1)
-# To plot:
-import matplotlib.pyplot as plt
+X = np.c_[np.ones((n,1)), x]
+XT_X = X.T @ X
+theta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)
+print("Own inversion")
+print(theta_linreg)
+# Hessian matrix
+H = (2.0/n)* XT_X
+EigValues, EigVectors = np.linalg.eig(H)
+print(f"Eigenvalues of Hessian Matrix:{EigValues}")
+
+theta = np.random.randn(2,1)
+eta = 1.0/np.max(EigValues)
+Niterations = 1000
-def f(x):
- return np.sin(2*np.pi*x + x**2)
+for iter in range(Niterations):
+ gradients = 2.0/n*X.T @ ((X @ theta)-y)
+ theta -= eta*gradients
+print("theta from own gd")
+print(theta)
-def f_grad_analytic(x):
- return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
+xnew = np.array([[0],[2]])
+Xnew = np.c_[np.ones((2,1)), xnew]
+ypredict = Xnew.dot(theta)
+ypredict2 = Xnew.dot(theta_linreg)
-# Do the comparison:
-x = np.linspace(0,1,1000)
+n_epochs = 50
+M = 5 #size of each minibatch
+m = int(n/M) #number of minibatches
+t0, t1 = 5, 50
+def learning_schedule(t):
+ return t0/(t+t1)
-f_grad = egrad(f)
+theta = np.random.randn(2,1)
-computed = f_grad(x)
-analytic = f_grad_analytic(x)
-
-plt.title('Derivative computed from Autograd compared with the analytical derivative')
-plt.plot(x,computed,label='autograd')
-plt.plot(x,analytic,label='analytic')
-
-plt.xlabel('x')
-plt.ylabel('y')
-plt.legend()
+for epoch in range(n_epochs):
+# Can you figure out a better way of setting up the contributions to each batch?
+ for i in range(m):
+ random_index = np.random.randint(m)
+ xi = X[random_index*M:random_index*M+M]
+ yi = y[random_index*M:random_index*M+M]
+ gradients = (2.0/M)* xi.T @ ((xi @ theta)-yi)
+ eta = learning_schedule(epoch*m+i)
+ theta = theta - eta*gradients
+print("theta from own sdg")
+print(theta)
+plt.plot(xnew, ypredict, "r-")
+plt.plot(xnew, ypredict2, "b-")
+plt.plot(x, y ,'ro')
+plt.axis([0,2.0,0, 15.0])
+plt.xlabel(r'$x$')
+plt.ylabel(r'$y$')
+plt.title(r'Random numbers ')
plt.show()
-
-print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
-
-Here we -experiment with what kind of functions Autograd is capable -of finding the gradient of. The following Python functions are just -meant to illustrate what Autograd can do, but please feel free to -experiment with other, possibly more complicated, functions as well. -
- - - -import autograd.numpy as np
-from autograd import grad
-
-def f1(x):
- return x**3 + 1
-
-f1_grad = grad(f1)
-
-# Remember to send in float as argument to the computed gradient from Autograd!
-a = 1.0
-
-# See the evaluated gradient at a using autograd:
-print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
-
-# Compare with the analytical derivative, that is f1'(x) = 3*x**2
-grad_analytical = 3*a**2
-print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
-
-To differentiate with respect to two (or more) arguments of a Python -function, Autograd need to know at which variable the function if -being differentiated with respect to. -
- - - -import autograd.numpy as np
-from autograd import grad
-def f2(x1,x2):
- return 3*x1**3 + x2*(x1 - 5) + 1
-
-# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
-f2_grad_x1 = grad(f2,0)
-
-# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
-f2_grad_x2 = grad(f2,1)
-
-x1 = 1.0
-x2 = 3.0
-
-print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
-print("-"*30)
-
-# Compare with the analytical derivatives:
-
-# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
-f2_grad_x1_analytical = 9*x1**2 + x2
-
-# Derivative of f2 w.r.t x2 is: x1 - 5:
-f2_grad_x2_analytical = x1 - 5
-
-# See the evaluated derivations:
-print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
-print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
-
-print()
-
-print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
-print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
-
-Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable.
-import autograd.numpy as np
-from autograd import grad
-def f3(x): # Assumes x is an array of length 5 or higher
- return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
-
-f3_grad = grad(f3)
-
-x = np.linspace(0,4,5)
-
-# Print the computed gradient:
-print("The computed gradient of f3 is: ", f3_grad(x))
-
-# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
-f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
-
-# Print the analytical gradient:
-print("The analytical gradient of f3 is: ", f3_grad_analytical)
-
-Note that in this case, when sending an array as input argument, the -output from Autograd is another array. This is the true gradient of -the function, as opposed to the function in the previous example. By -using arrays to represent the variables, the output from Autograd -might be easier to work with, as the output is closer to what one -could expect form a gradient-evaluting function. -
-import autograd.numpy as np
-from autograd import grad
-def f4(x):
- return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
-
-f4_grad = grad(f4)
-
-x = 2.7
-
-# Print the computed derivative:
-print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
-
-# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
-f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
-
-# Print the analytical gradient:
-print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
-
-import autograd.numpy as np
-from autograd import grad
-def f5(x):
- if x >= 0:
- return x**2
- else:
- return -3*x + 1
-
-f5_grad = grad(f5)
-
-x = 2.7
-
-# Print the computed derivative:
-print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
-
-import autograd.numpy as np
-from autograd import grad
-def f6_for(x):
- val = 0
- for i in range(10):
- val = val + x**i
- return val
-
-def f6_while(x):
- val = 0
- i = 0
- while i < 10:
- val = val + x**i
- i = i + 1
- return val
-
-f6_for_grad = grad(f6_for)
-f6_while_grad = grad(f6_while)
-
-x = 0.5
-
-# Print the computed derivaties of f6_for and f6_while
-print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
-print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
-
-import autograd.numpy as np
-from autograd import grad
-# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
-# The analytical derivative is: sum(i*x**(i-1))
-f6_grad_analytical = 0
-for i in range(10):
- f6_grad_analytical += i*x**(i-1)
-
-print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
-
-import autograd.numpy as np
-from autograd import grad
-
-def f7(n): # Assume that n is an integer
- if n == 1 or n == 0:
- return 1
- else:
- return n*f7(n-1)
-
-f7_grad = grad(f7)
-
-n = 2.0
-
-print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
-
-# The function f7 is an implementation of the factorial of n.
-# By using the product rule, one can find that the derivative is:
-
-f7_grad_analytical = 0
-for i in range(int(n)-1):
- tmp = 1
- for k in range(int(n)-1):
- if k != i:
- tmp *= (n - k)
- f7_grad_analytical += tmp
-
-print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
-
-Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input.
-Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd.
- -Assigning a value to the variable being differentiated with respect to
- - -import autograd.numpy as np
-from autograd import grad
-def f8(x): # Assume x is an array
- x[2] = 3
- return x*2
-
-f8_grad = grad(f8)
-
-x = 8.4
-
-print("The derivative of f8 is:",f8_grad(x))
-
-Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible.
-import autograd.numpy as np
-from autograd import grad
-def f9(a): # Assume a is an array with 2 elements
- b = np.array([1.0,2.0])
- return a.dot(b)
-
-f9_grad = grad(f9)
-
-x = np.array([1.0,0.0])
-
-print("The derivative of f9 is:",f9_grad(x))
-
-Here we are told that the 'dot' function does not belong to Autograd's -version of a Numpy array. To overcome this, an alternative syntax -which also computed the dot product can be used: -
- - - -import autograd.numpy as np
-from autograd import grad
-def f9_alternative(x): # Assume a is an array with 2 elements
- b = np.array([1.0,2.0])
- return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
-
-f9_alternative_grad = grad(f9_alternative)
-
-x = np.array([3.0,0.0])
-
-print("The gradient of f9 is:",f9_alternative_grad(x))
-
-# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
-# w.r.t x is (b_1, b_2).
-
-The documentation recommends to avoid inplace operations such as
- - -a += b
-a -= b
-a*= b
-a /=b
Challenge: try to write a similar code for a Logistic Regression case.
The stochastic gradient descent (SGD) is almost always used with a -momentum or inertia term that serves as a memory of the direction we -are moving in parameter space. This is typically implemented as -follows -
+In the code here we vary the number of mini-batches.
-$$ -\begin{align} -\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t) \nonumber \\ -\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t}, -\label{_auto1} -\end{align} -$$ - -where we have introduced a momentum parameter \( \gamma \), with -\( 0\le\gamma\le 1 \), and for brevity we dropped the explicit notation to -indicate the gradient is to be taken over a different mini-batch at -each step. We call this algorithm gradient descent with momentum -(GDM). From these equations, it is clear that \( \mathbf{v}_t \) is a -running average of recently encountered gradients and -\( (1-\gamma)^{-1} \) sets the characteristic time scale for the memory -used in the averaging procedure. Consistent with this, when -\( \gamma=0 \), this just reduces down to ordinary SGD as discussed -earlier. An equivalent way of writing the updates is -
- -$$ -\Delta \boldsymbol{\theta}_{t+1} = \gamma \Delta \boldsymbol{\theta}_t -\ \eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t), -$$ - -where we have defined \( \Delta \boldsymbol{\theta}_{t}= \boldsymbol{\theta}_t-\boldsymbol{\theta}_{t-1} \).
- -Let us try to get more intuition from these equations. It is helpful -to consider a simple physical analogy with a particle of mass \( m \) -moving in a viscous medium with drag coefficient \( \mu \) and potential -\( E(\mathbf{w}) \). If we denote the particle's position by \( \mathbf{w} \), -then its motion is described by -
- -$$ -m {d^2 \mathbf{w} \over dt^2} + \mu {d \mathbf{w} \over dt }= -\nabla_w E(\mathbf{w}). -$$ - -We can discretize this equation in the usual way to get
- -$$ -m { \mathbf{w}_{t+\Delta t}-2 \mathbf{w}_{t} +\mathbf{w}_{t-\Delta t} \over (\Delta t)^2}+\mu {\mathbf{w}_{t+\Delta t}- \mathbf{w}_{t} \over \Delta t} = -\nabla_w E(\mathbf{w}). -$$ - -Rearranging this equation, we can rewrite this as
- -$$ -\Delta \mathbf{w}_{t +\Delta t}= - { (\Delta t)^2 \over m +\mu \Delta t} \nabla_w E(\mathbf{w})+ {m \over m +\mu \Delta t} \Delta \mathbf{w}_t. -$$ - - -Notice that this equation is identical to previous one if we identify -the position of the particle, \( \mathbf{w} \), with the parameters -\( \boldsymbol{\theta} \). This allows us to identify the momentum -parameter and learning rate with the mass of the particle and the -viscous drag as: -
- -$$ -\gamma= {m \over m +\mu \Delta t }, \qquad \eta = {(\Delta t)^2 \over m +\mu \Delta t}. -$$ - -Thus, as the name suggests, the momentum parameter is proportional to -the mass of the particle and effectively provides inertia. -Furthermore, in the large viscosity/small learning rate limit, our -memory time scales as \( (1-\gamma)^{-1} \approx m/(\mu \Delta t) \). -
- -Why is momentum useful? SGD momentum helps the gradient descent -algorithm gain speed in directions with persistent but small gradients -even in the presence of stochasticity, while suppressing oscillations -in high-curvature directions. This becomes especially important in -situations where the landscape is shallow and flat in some directions -and narrow and steep in others. It has been argued that first-order -methods (with appropriate initial conditions) can perform comparable -to more expensive second order methods, especially in the context of -complex deep learning models. -
- -These beneficial properties of momentum can sometimes become even more -pronounced by using a slight modification of the classical momentum -algorithm called Nesterov Accelerated Gradient (NAG). -
- -In the NAG algorithm, rather than calculating the gradient at the -current parameters, \( \nabla_\theta E(\boldsymbol{\theta}_t) \), one -calculates the gradient at the expected value of the parameters given -our current momentum, \( \nabla_\theta E(\boldsymbol{\theta}_t +\gamma -\mathbf{v}_{t-1}) \). This yields the NAG update rule -
- -$$ -\begin{align} -\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t +\gamma \mathbf{v}_{t-1}) \nonumber \\ -\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t}. -\label{_auto2} -\end{align} -$$ - -One of the major advantages of NAG is that it allows for the use of a larger learning rate than GDM for the same choice of \( \gamma \).
- -In stochastic gradient descent, with and without momentum, we still -have to specify a schedule for tuning the learning rates \( \eta_t \) -as a function of time. As discussed in the context of Newton's -method, this presents a number of dilemmas. The learning rate is -limited by the steepest direction which can change depending on the -current position in the landscape. To circumvent this problem, ideally -our algorithm would keep track of curvature and take large steps in -shallow, flat directions and small steps in steep, narrow directions. -Second-order methods accomplish this by calculating or approximating -the Hessian and normalizing the learning rate by the -curvature. However, this is very computationally expensive for -extremely large models. Ideally, we would like to be able to -adaptively change the step size to match the landscape without paying -the steep computational price of calculating or approximating -Hessians. -
- -Recently, a number of methods have been introduced that accomplish -this by tracking not only the gradient, but also the second moment of -the gradient. These methods include AdaGrad, AdaDelta, RMS-Prop, and -ADAM. -
- -In RMS prop, in addition to keeping a running average of the first -moment of the gradient, we also keep track of the second moment -denoted by \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}_t^2] \). The update rule -for RMS prop is given by -
- -$$ -\begin{align} -\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta}) -\label{_auto3}\\ -\mathbf{s}_t &=\beta \mathbf{s}_{t-1} +(1-\beta)\mathbf{g}_t^2 \nonumber \\ -\boldsymbol{\theta}_{t+1}&=&\boldsymbol{\theta}_t - \eta_t { \mathbf{g}_t \over \sqrt{\mathbf{s}_t +\epsilon}}, \nonumber -\end{align} -$$ - -where \( \beta \) controls the averaging time of the second moment and is -typically taken to be about \( \beta=0.9 \), \( \eta_t \) is a learning rate -typically chosen to be \( 10^{-3} \), and \( \epsilon\sim 10^{-8} \) is a -small regularization constant to prevent divergences. Multiplication -and division by vectors is understood as an element-wise operation. It -is clear from this formula that the learning rate is reduced in -directions where the norm of the gradient is consistently large. This -greatly speeds up the convergence by allowing us to use a larger -learning rate for flat directions. -
- -A related algorithm is the ADAM optimizer. In ADAM, we keep a running -average of both the first and second moment of the gradient and use -this information to adaptively change the learning rate for different -parameters. In addition to keeping a running average of the first and -second moments of the gradient -(i.e. \( \mathbf{m}_t=\mathbb{E}[\mathbf{g}_t] \) and -\( \mathbf{s}_t=\mathbb{E}[\mathbf{g}^2_t] \), respectively), ADAM -performs an additional bias correction to account for the fact that we -are estimating the first two moments of the gradient using a running -average (denoted by the hats in the update rule below). The update -rule for ADAM is given by (where multiplication and division are once -again understood to be element-wise operations below) -
- -$$ -\begin{align} -\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta}) -\label{_auto4}\\ -\mathbf{m}_t &= \beta_1 \mathbf{m}_{t-1} + (1-\beta_1) \mathbf{g}_t \nonumber \\ -\mathbf{s}_t &=\beta_2 \mathbf{s}_{t-1} +(1-\beta_2)\mathbf{g}_t^2 \nonumber \\ -\boldsymbol{\mathbf{m}}_t&={\mathbf{m}_t \over 1-\beta_1^t} \nonumber \\ -\boldsymbol{\mathbf{s}}_t &={\mathbf{s}_t \over1-\beta_2^t} \nonumber \\ -\boldsymbol{\theta}_{t+1}&=\boldsymbol{\theta}_t - \eta_t { \boldsymbol{\mathbf{m}}_t \over \sqrt{\boldsymbol{\mathbf{s}}_t} +\epsilon}, \nonumber \\ -\label{_auto5} -\end{align} -$$ - -where \( \beta_1 \) and \( \beta_2 \) set the memory lifetime of the first and -second moment and are typically taken to be \( 0.9 \) and \( 0.99 \) -respectively, and \( \eta \) and \( \epsilon \) are identical to RMSprop. -
- -Like in RMSprop, the effective step size of a parameter depends on the -magnitude of its gradient squared. To understand this better, let us -rewrite this expression in terms of the variance -\( \boldsymbol{\sigma}_t^2 = \boldsymbol{\mathbf{s}}_t - -(\boldsymbol{\mathbf{m}}_t)^2 \). Consider a single parameter \( \theta_t \). The -update rule for this parameter is given by -
- -$$ -\Delta \theta_{t+1}= -\eta_t { \boldsymbol{m}_t \over \sqrt{\sigma_t^2 + m_t^2 }+\epsilon}. -$$ - - -Geron's text, see chapter 11, has several interesting discussions.
- -Automatic differentiation (AD), -also called algorithmic -differentiation or computational differentiation,is a set of -techniques to numerically evaluate the derivative of a function -specified by a computer program. AD exploits the fact that every -computer program, no matter how complicated, executes a sequence of -elementary arithmetic operations (addition, subtraction, -multiplication, division, etc.) and elementary functions (exp, log, -sin, cos, etc.). By applying the chain rule repeatedly to these -operations, derivatives of arbitrary order can be computed -automatically, accurately to working precision, and using at most a -small constant factor more arithmetic operations than the original -program. -
- -Automatic differentiation is neither:
- -Symbolic differentiation can lead to inefficient code and faces the -difficulty of converting a computer program into a single expression, -while numerical differentiation can introduce round-off errors in the -discretization process and cancellation -
- -Python has tools for so-called automatic differentiation. -Consider the following example -
-$$ -f(x) = \sin\left(2\pi x + x^2\right) -$$ - -which has the following derivative
-$$ -f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) -$$ - -Using autograd we have
- - - +import autograd.numpy as np
+ # Importing various packages
+from math import exp, sqrt
+from random import random, seed
+import numpy as np
+import matplotlib.pyplot as plt
-# To do elementwise differentiation:
-from autograd import elementwise_grad as egrad
+n = 100
+x = 2*np.random.rand(n,1)
+y = 4+3*x+np.random.randn(n,1)
-# To plot:
-import matplotlib.pyplot as plt
+X = np.c_[np.ones((n,1)), x]
+XT_X = X.T @ X
+theta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)
+print("Own inversion")
+print(theta_linreg)
+# Hessian matrix
+H = (2.0/n)* XT_X
+EigValues, EigVectors = np.linalg.eig(H)
+print(f"Eigenvalues of Hessian Matrix:{EigValues}")
+
+theta = np.random.randn(2,1)
+eta = 1.0/np.max(EigValues)
+Niterations = 1000
-def f(x):
- return np.sin(2*np.pi*x + x**2)
+for iter in range(Niterations):
+ gradients = 2.0/n*X.T @ ((X @ theta)-y)
+ theta -= eta*gradients
+print("theta from own gd")
+print(theta)
-def f_grad_analytic(x):
- return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
+xnew = np.array([[0],[2]])
+Xnew = np.c_[np.ones((2,1)), xnew]
+ypredict = Xnew.dot(theta)
+ypredict2 = Xnew.dot(theta_linreg)
-# Do the comparison:
-x = np.linspace(0,1,1000)
+n_epochs = 50
+M = 5 #size of each minibatch
+m = int(n/M) #number of minibatches
+t0, t1 = 5, 50
+def learning_schedule(t):
+ return t0/(t+t1)
-f_grad = egrad(f)
+theta = np.random.randn(2,1)
-computed = f_grad(x)
-analytic = f_grad_analytic(x)
-
-plt.title('Derivative computed from Autograd compared with the analytical derivative')
-plt.plot(x,computed,label='autograd')
-plt.plot(x,analytic,label='analytic')
-
-plt.xlabel('x')
-plt.ylabel('y')
-plt.legend()
+for epoch in range(n_epochs):
+# Can you figure out a better way of setting up the contributions to each batch?
+ for i in range(m):
+ random_index = np.random.randint(m)
+ xi = X[random_index*M:random_index*M+M]
+ yi = y[random_index*M:random_index*M+M]
+ gradients = (2.0/M)* xi.T @ ((xi @ theta)-yi)
+ eta = learning_schedule(epoch*m+i)
+ theta = theta - eta*gradients
+print("theta from own sdg")
+print(theta)
+plt.plot(xnew, ypredict, "r-")
+plt.plot(xnew, ypredict2, "b-")
+plt.plot(x, y ,'ro')
+plt.axis([0,2.0,0, 15.0])
+plt.xlabel(r'$x$')
+plt.ylabel(r'$y$')
+plt.title(r'Random numbers ')
plt.show()
-
-print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
-
-Here we -experiment with what kind of functions Autograd is capable -of finding the gradient of. The following Python functions are just -meant to illustrate what Autograd can do, but please feel free to -experiment with other, possibly more complicated, functions as well. -
- - - -import autograd.numpy as np
-from autograd import grad
-
-def f1(x):
- return x**3 + 1
-
-f1_grad = grad(f1)
-
-# Remember to send in float as argument to the computed gradient from Autograd!
-a = 1.0
-
-# See the evaluated gradient at a using autograd:
-print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
-
-# Compare with the analytical derivative, that is f1'(x) = 3*x**2
-grad_analytical = 3*a**2
-print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
-
-To differentiate with respect to two (or more) arguments of a Python -function, Autograd need to know at which variable the function if -being differentiated with respect to. -
- - - -import autograd.numpy as np
-from autograd import grad
-def f2(x1,x2):
- return 3*x1**3 + x2*(x1 - 5) + 1
-
-# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
-f2_grad_x1 = grad(f2,0)
-
-# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
-f2_grad_x2 = grad(f2,1)
-
-x1 = 1.0
-x2 = 3.0
-
-print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
-print("-"*30)
-
-# Compare with the analytical derivatives:
-
-# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
-f2_grad_x1_analytical = 9*x1**2 + x2
-
-# Derivative of f2 w.r.t x2 is: x1 - 5:
-f2_grad_x2_analytical = x1 - 5
-
-# See the evaluated derivations:
-print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
-print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
-
-print()
-
-print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
-print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
-
-Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable.
- -import autograd.numpy as np
-from autograd import grad
-def f3(x): # Assumes x is an array of length 5 or higher
- return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
-
-f3_grad = grad(f3)
-
-x = np.linspace(0,4,5)
-
-# Print the computed gradient:
-print("The computed gradient of f3 is: ", f3_grad(x))
-
-# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
-f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
-
-# Print the analytical gradient:
-print("The analytical gradient of f3 is: ", f3_grad_analytical)
-
-Note that in this case, when sending an array as input argument, the -output from Autograd is another array. This is the true gradient of -the function, as opposed to the function in the previous example. By -using arrays to represent the variables, the output from Autograd -might be easier to work with, as the output is closer to what one -could expect form a gradient-evaluting function. -
- - -import autograd.numpy as np
-from autograd import grad
-def f4(x):
- return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
-
-f4_grad = grad(f4)
-
-x = 2.7
-
-# Print the computed derivative:
-print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
-
-# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
-f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
-
-# Print the analytical gradient:
-print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
-
-import autograd.numpy as np
-from autograd import grad
-def f5(x):
- if x >= 0:
- return x**2
- else:
- return -3*x + 1
-
-f5_grad = grad(f5)
-
-x = 2.7
-
-# Print the computed derivative:
-print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
-
-import autograd.numpy as np
-from autograd import grad
-def f6_for(x):
- val = 0
- for i in range(10):
- val = val + x**i
- return val
-
-def f6_while(x):
- val = 0
- i = 0
- while i < 10:
- val = val + x**i
- i = i + 1
- return val
-
-f6_for_grad = grad(f6_for)
-f6_while_grad = grad(f6_while)
-
-x = 0.5
-
-# Print the computed derivaties of f6_for and f6_while
-print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
-print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
-
-import autograd.numpy as np
-from autograd import grad
-# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
-# The analytical derivative is: sum(i*x**(i-1))
-f6_grad_analytical = 0
-for i in range(10):
- f6_grad_analytical += i*x**(i-1)
-
-print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
-
-import autograd.numpy as np
-from autograd import grad
-
-def f7(n): # Assume that n is an integer
- if n == 1 or n == 0:
- return 1
- else:
- return n*f7(n-1)
-
-f7_grad = grad(f7)
-
-n = 2.0
-
-print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
-
-# The function f7 is an implementation of the factorial of n.
-# By using the product rule, one can find that the derivative is:
-
-f7_grad_analytical = 0
-for i in range(int(n)-1):
- tmp = 1
- for k in range(int(n)-1):
- if k != i:
- tmp *= (n - k)
- f7_grad_analytical += tmp
-
-print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
-
-Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input.
- -Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd.
- -Assigning a value to the variable being differentiated with respect to
- - -import autograd.numpy as np
-from autograd import grad
-def f8(x): # Assume x is an array
- x[2] = 3
- return x*2
-
-f8_grad = grad(f8)
-
-x = 8.4
-
-print("The derivative of f8 is:",f8_grad(x))
-
-Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible.
- -import autograd.numpy as np
-from autograd import grad
-def f9(a): # Assume a is an array with 2 elements
- b = np.array([1.0,2.0])
- return a.dot(b)
-
-f9_grad = grad(f9)
-
-x = np.array([1.0,0.0])
-
-print("The derivative of f9 is:",f9_grad(x))
-
-Here we are told that the 'dot' function does not belong to Autograd's -version of a Numpy array. To overcome this, an alternative syntax -which also computed the dot product can be used: -
- - - -import autograd.numpy as np
-from autograd import grad
-def f9_alternative(x): # Assume a is an array with 2 elements
- b = np.array([1.0,2.0])
- return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
-
-f9_alternative_grad = grad(f9_alternative)
-
-x = np.array([3.0,0.0])
-
-print("The gradient of f9 is:",f9_alternative_grad(x))
-
-# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
-# w.r.t x is (b_1, b_2).
-
-The documentation recommends to avoid inplace operations such as
- - -a += b
-a -= b
-a*= b
-a /=b
Challenge: try to write a similar code for a Logistic Regression case.
The stochastic gradient descent (SGD) is almost always used with a -momentum or inertia term that serves as a memory of the direction we -are moving in parameter space. This is typically implemented as -follows -
+In the code here we vary the number of mini-batches.
-$$ -\begin{align} -\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t) \nonumber \\ -\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t}, -\label{_auto1} -\end{align} -$$ - -where we have introduced a momentum parameter \( \gamma \), with -\( 0\le\gamma\le 1 \), and for brevity we dropped the explicit notation to -indicate the gradient is to be taken over a different mini-batch at -each step. We call this algorithm gradient descent with momentum -(GDM). From these equations, it is clear that \( \mathbf{v}_t \) is a -running average of recently encountered gradients and -\( (1-\gamma)^{-1} \) sets the characteristic time scale for the memory -used in the averaging procedure. Consistent with this, when -\( \gamma=0 \), this just reduces down to ordinary SGD as discussed -earlier. An equivalent way of writing the updates is -
- -$$ -\Delta \boldsymbol{\theta}_{t+1} = \gamma \Delta \boldsymbol{\theta}_t -\ \eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t), -$$ - -where we have defined \( \Delta \boldsymbol{\theta}_{t}= \boldsymbol{\theta}_t-\boldsymbol{\theta}_{t-1} \).
- -Let us try to get more intuition from these equations. It is helpful -to consider a simple physical analogy with a particle of mass \( m \) -moving in a viscous medium with drag coefficient \( \mu \) and potential -\( E(\mathbf{w}) \). If we denote the particle's position by \( \mathbf{w} \), -then its motion is described by -
- -$$ -m {d^2 \mathbf{w} \over dt^2} + \mu {d \mathbf{w} \over dt }= -\nabla_w E(\mathbf{w}). -$$ - -We can discretize this equation in the usual way to get
- -$$ -m { \mathbf{w}_{t+\Delta t}-2 \mathbf{w}_{t} +\mathbf{w}_{t-\Delta t} \over (\Delta t)^2}+\mu {\mathbf{w}_{t+\Delta t}- \mathbf{w}_{t} \over \Delta t} = -\nabla_w E(\mathbf{w}). -$$ - -Rearranging this equation, we can rewrite this as
- -$$ -\Delta \mathbf{w}_{t +\Delta t}= - { (\Delta t)^2 \over m +\mu \Delta t} \nabla_w E(\mathbf{w})+ {m \over m +\mu \Delta t} \Delta \mathbf{w}_t. -$$ - - -Notice that this equation is identical to previous one if we identify -the position of the particle, \( \mathbf{w} \), with the parameters -\( \boldsymbol{\theta} \). This allows us to identify the momentum -parameter and learning rate with the mass of the particle and the -viscous drag as: -
- -$$ -\gamma= {m \over m +\mu \Delta t }, \qquad \eta = {(\Delta t)^2 \over m +\mu \Delta t}. -$$ - -Thus, as the name suggests, the momentum parameter is proportional to -the mass of the particle and effectively provides inertia. -Furthermore, in the large viscosity/small learning rate limit, our -memory time scales as \( (1-\gamma)^{-1} \approx m/(\mu \Delta t) \). -
- -Why is momentum useful? SGD momentum helps the gradient descent -algorithm gain speed in directions with persistent but small gradients -even in the presence of stochasticity, while suppressing oscillations -in high-curvature directions. This becomes especially important in -situations where the landscape is shallow and flat in some directions -and narrow and steep in others. It has been argued that first-order -methods (with appropriate initial conditions) can perform comparable -to more expensive second order methods, especially in the context of -complex deep learning models. -
- -These beneficial properties of momentum can sometimes become even more -pronounced by using a slight modification of the classical momentum -algorithm called Nesterov Accelerated Gradient (NAG). -
- -In the NAG algorithm, rather than calculating the gradient at the -current parameters, \( \nabla_\theta E(\boldsymbol{\theta}_t) \), one -calculates the gradient at the expected value of the parameters given -our current momentum, \( \nabla_\theta E(\boldsymbol{\theta}_t +\gamma -\mathbf{v}_{t-1}) \). This yields the NAG update rule -
- -$$ -\begin{align} -\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t +\gamma \mathbf{v}_{t-1}) \nonumber \\ -\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t}. -\label{_auto2} -\end{align} -$$ - -One of the major advantages of NAG is that it allows for the use of a larger learning rate than GDM for the same choice of \( \gamma \).
- -In stochastic gradient descent, with and without momentum, we still -have to specify a schedule for tuning the learning rates \( \eta_t \) -as a function of time. As discussed in the context of Newton's -method, this presents a number of dilemmas. The learning rate is -limited by the steepest direction which can change depending on the -current position in the landscape. To circumvent this problem, ideally -our algorithm would keep track of curvature and take large steps in -shallow, flat directions and small steps in steep, narrow directions. -Second-order methods accomplish this by calculating or approximating -the Hessian and normalizing the learning rate by the -curvature. However, this is very computationally expensive for -extremely large models. Ideally, we would like to be able to -adaptively change the step size to match the landscape without paying -the steep computational price of calculating or approximating -Hessians. -
- -Recently, a number of methods have been introduced that accomplish -this by tracking not only the gradient, but also the second moment of -the gradient. These methods include AdaGrad, AdaDelta, RMS-Prop, and -ADAM. -
- -In RMS prop, in addition to keeping a running average of the first -moment of the gradient, we also keep track of the second moment -denoted by \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}_t^2] \). The update rule -for RMS prop is given by -
- -$$ -\begin{align} -\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta}) -\label{_auto3}\\ -\mathbf{s}_t &=\beta \mathbf{s}_{t-1} +(1-\beta)\mathbf{g}_t^2 \nonumber \\ -\boldsymbol{\theta}_{t+1}&=&\boldsymbol{\theta}_t - \eta_t { \mathbf{g}_t \over \sqrt{\mathbf{s}_t +\epsilon}}, \nonumber -\end{align} -$$ - -where \( \beta \) controls the averaging time of the second moment and is -typically taken to be about \( \beta=0.9 \), \( \eta_t \) is a learning rate -typically chosen to be \( 10^{-3} \), and \( \epsilon\sim 10^{-8} \) is a -small regularization constant to prevent divergences. Multiplication -and division by vectors is understood as an element-wise operation. It -is clear from this formula that the learning rate is reduced in -directions where the norm of the gradient is consistently large. This -greatly speeds up the convergence by allowing us to use a larger -learning rate for flat directions. -
- -A related algorithm is the ADAM optimizer. In ADAM, we keep a running -average of both the first and second moment of the gradient and use -this information to adaptively change the learning rate for different -parameters. In addition to keeping a running average of the first and -second moments of the gradient -(i.e. \( \mathbf{m}_t=\mathbb{E}[\mathbf{g}_t] \) and -\( \mathbf{s}_t=\mathbb{E}[\mathbf{g}^2_t] \), respectively), ADAM -performs an additional bias correction to account for the fact that we -are estimating the first two moments of the gradient using a running -average (denoted by the hats in the update rule below). The update -rule for ADAM is given by (where multiplication and division are once -again understood to be element-wise operations below) -
- -$$ -\begin{align} -\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta}) -\label{_auto4}\\ -\mathbf{m}_t &= \beta_1 \mathbf{m}_{t-1} + (1-\beta_1) \mathbf{g}_t \nonumber \\ -\mathbf{s}_t &=\beta_2 \mathbf{s}_{t-1} +(1-\beta_2)\mathbf{g}_t^2 \nonumber \\ -\boldsymbol{\mathbf{m}}_t&={\mathbf{m}_t \over 1-\beta_1^t} \nonumber \\ -\boldsymbol{\mathbf{s}}_t &={\mathbf{s}_t \over1-\beta_2^t} \nonumber \\ -\boldsymbol{\theta}_{t+1}&=\boldsymbol{\theta}_t - \eta_t { \boldsymbol{\mathbf{m}}_t \over \sqrt{\boldsymbol{\mathbf{s}}_t} +\epsilon}, \nonumber \\ -\label{_auto5} -\end{align} -$$ - -where \( \beta_1 \) and \( \beta_2 \) set the memory lifetime of the first and -second moment and are typically taken to be \( 0.9 \) and \( 0.99 \) -respectively, and \( \eta \) and \( \epsilon \) are identical to RMSprop. -
- -Like in RMSprop, the effective step size of a parameter depends on the -magnitude of its gradient squared. To understand this better, let us -rewrite this expression in terms of the variance -\( \boldsymbol{\sigma}_t^2 = \boldsymbol{\mathbf{s}}_t - -(\boldsymbol{\mathbf{m}}_t)^2 \). Consider a single parameter \( \theta_t \). The -update rule for this parameter is given by -
- -$$ -\Delta \theta_{t+1}= -\eta_t { \boldsymbol{m}_t \over \sqrt{\sigma_t^2 + m_t^2 }+\epsilon}. -$$ - - -Geron's text, see chapter 11, has several interesting discussions.
- -Automatic differentiation (AD), -also called algorithmic -differentiation or computational differentiation,is a set of -techniques to numerically evaluate the derivative of a function -specified by a computer program. AD exploits the fact that every -computer program, no matter how complicated, executes a sequence of -elementary arithmetic operations (addition, subtraction, -multiplication, division, etc.) and elementary functions (exp, log, -sin, cos, etc.). By applying the chain rule repeatedly to these -operations, derivatives of arbitrary order can be computed -automatically, accurately to working precision, and using at most a -small constant factor more arithmetic operations than the original -program. -
- -Automatic differentiation is neither:
- -Symbolic differentiation can lead to inefficient code and faces the -difficulty of converting a computer program into a single expression, -while numerical differentiation can introduce round-off errors in the -discretization process and cancellation -
- -Python has tools for so-called automatic differentiation. -Consider the following example -
-$$ -f(x) = \sin\left(2\pi x + x^2\right) -$$ - -which has the following derivative
-$$ -f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) -$$ - -Using autograd we have
- - - +import autograd.numpy as np
-
-# To do elementwise differentiation:
-from autograd import elementwise_grad as egrad
-
-# To plot:
-import matplotlib.pyplot as plt
-
-
-def f(x):
- return np.sin(2*np.pi*x + x**2)
-
-def f_grad_analytic(x):
- return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
-
-# Do the comparison:
-x = np.linspace(0,1,1000)
-
-f_grad = egrad(f)
-
-computed = f_grad(x)
-analytic = f_grad_analytic(x)
-
-plt.title('Derivative computed from Autograd compared with the analytical derivative')
-plt.plot(x,computed,label='autograd')
-plt.plot(x,analytic,label='analytic')
-
-plt.xlabel('x')
-plt.ylabel('y')
-plt.legend()
-
-plt.show()
-
-print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
-
-Here we -experiment with what kind of functions Autograd is capable -of finding the gradient of. The following Python functions are just -meant to illustrate what Autograd can do, but please feel free to -experiment with other, possibly more complicated, functions as well. -
- - - -import autograd.numpy as np
-from autograd import grad
-
-def f1(x):
- return x**3 + 1
-
-f1_grad = grad(f1)
-
-# Remember to send in float as argument to the computed gradient from Autograd!
-a = 1.0
-
-# See the evaluated gradient at a using autograd:
-print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
-
-# Compare with the analytical derivative, that is f1'(x) = 3*x**2
-grad_analytical = 3*a**2
-print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
-
-To differentiate with respect to two (or more) arguments of a Python -function, Autograd need to know at which variable the function if -being differentiated with respect to. -
- - - -import autograd.numpy as np
-from autograd import grad
-def f2(x1,x2):
- return 3*x1**3 + x2*(x1 - 5) + 1
-
-# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
-f2_grad_x1 = grad(f2,0)
-
-# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
-f2_grad_x2 = grad(f2,1)
-
-x1 = 1.0
-x2 = 3.0
-
-print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
-print("-"*30)
-
-# Compare with the analytical derivatives:
-
-# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
-f2_grad_x1_analytical = 9*x1**2 + x2
-
-# Derivative of f2 w.r.t x2 is: x1 - 5:
-f2_grad_x2_analytical = x1 - 5
-
-# See the evaluated derivations:
-print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
-print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
-
-print()
-
-print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
-print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
-
-Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable.
- -import autograd.numpy as np
-from autograd import grad
-def f3(x): # Assumes x is an array of length 5 or higher
- return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
-
-f3_grad = grad(f3)
-
-x = np.linspace(0,4,5)
-
-# Print the computed gradient:
-print("The computed gradient of f3 is: ", f3_grad(x))
-
-# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
-f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
-
-# Print the analytical gradient:
-print("The analytical gradient of f3 is: ", f3_grad_analytical)
-
-Note that in this case, when sending an array as input argument, the -output from Autograd is another array. This is the true gradient of -the function, as opposed to the function in the previous example. By -using arrays to represent the variables, the output from Autograd -might be easier to work with, as the output is closer to what one -could expect form a gradient-evaluting function. -
- - -import autograd.numpy as np
-from autograd import grad
-def f4(x):
- return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
-
-f4_grad = grad(f4)
-
-x = 2.7
-
-# Print the computed derivative:
-print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
-
-# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
-f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
-
-# Print the analytical gradient:
-print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
-
-import autograd.numpy as np
-from autograd import grad
-def f5(x):
- if x >= 0:
- return x**2
- else:
- return -3*x + 1
-
-f5_grad = grad(f5)
-
-x = 2.7
-
-# Print the computed derivative:
-print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
-
-import autograd.numpy as np
-from autograd import grad
-def f6_for(x):
- val = 0
- for i in range(10):
- val = val + x**i
- return val
-
-def f6_while(x):
- val = 0
- i = 0
- while i < 10:
- val = val + x**i
- i = i + 1
- return val
-
-f6_for_grad = grad(f6_for)
-f6_while_grad = grad(f6_while)
-
-x = 0.5
-
-# Print the computed derivaties of f6_for and f6_while
-print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
-print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
-
-import autograd.numpy as np
-from autograd import grad
-# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
-# The analytical derivative is: sum(i*x**(i-1))
-f6_grad_analytical = 0
-for i in range(10):
- f6_grad_analytical += i*x**(i-1)
-
-print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
-
-import autograd.numpy as np
-from autograd import grad
-
-def f7(n): # Assume that n is an integer
- if n == 1 or n == 0:
- return 1
- else:
- return n*f7(n-1)
-
-f7_grad = grad(f7)
-
-n = 2.0
-
-print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
-
-# The function f7 is an implementation of the factorial of n.
-# By using the product rule, one can find that the derivative is:
-
-f7_grad_analytical = 0
-for i in range(int(n)-1):
- tmp = 1
- for k in range(int(n)-1):
- if k != i:
- tmp *= (n - k)
- f7_grad_analytical += tmp
-
-print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
-
-Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input.
- -Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd.
- -Assigning a value to the variable being differentiated with respect to
- - -import autograd.numpy as np
-from autograd import grad
-def f8(x): # Assume x is an array
- x[2] = 3
- return x*2
-
-f8_grad = grad(f8)
-
-x = 8.4
-
-print("The derivative of f8 is:",f8_grad(x))
-
-Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible.
- -import autograd.numpy as np
-from autograd import grad
-def f9(a): # Assume a is an array with 2 elements
- b = np.array([1.0,2.0])
- return a.dot(b)
-
-f9_grad = grad(f9)
-
-x = np.array([1.0,0.0])
-
-print("The derivative of f9 is:",f9_grad(x))
-
-Here we are told that the 'dot' function does not belong to Autograd's -version of a Numpy array. To overcome this, an alternative syntax -which also computed the dot product can be used: -
- - - -import autograd.numpy as np
-from autograd import grad
-def f9_alternative(x): # Assume a is an array with 2 elements
- b = np.array([1.0,2.0])
- return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
-
-f9_alternative_grad = grad(f9_alternative)
-
-x = np.array([3.0,0.0])
-
-print("The gradient of f9 is:",f9_alternative_grad(x))
-
-# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
-# w.r.t x is (b_1, b_2).
-
-The documentation recommends to avoid inplace operations such as
- - -a += b
-a -= b
-a*= b
-a /=b
+ # Importing various packages
+from math import exp, sqrt
+from random import random, seed
+import numpy as np
+import matplotlib.pyplot as plt
+
+n = 100
+x = 2*np.random.rand(n,1)
+y = 4+3*x+np.random.randn(n,1)
+
+X = np.c_[np.ones((n,1)), x]
+XT_X = X.T @ X
+theta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)
+print("Own inversion")
+print(theta_linreg)
+# Hessian matrix
+H = (2.0/n)* XT_X
+EigValues, EigVectors = np.linalg.eig(H)
+print(f"Eigenvalues of Hessian Matrix:{EigValues}")
+
+theta = np.random.randn(2,1)
+eta = 1.0/np.max(EigValues)
+Niterations = 1000
+
+
+for iter in range(Niterations):
+ gradients = 2.0/n*X.T @ ((X @ theta)-y)
+ theta -= eta*gradients
+print("theta from own gd")
+print(theta)
+
+xnew = np.array([[0],[2]])
+Xnew = np.c_[np.ones((2,1)), xnew]
+ypredict = Xnew.dot(theta)
+ypredict2 = Xnew.dot(theta_linreg)
+
+n_epochs = 50
+M = 5 #size of each minibatch
+m = int(n/M) #number of minibatches
+t0, t1 = 5, 50
+def learning_schedule(t):
+ return t0/(t+t1)
+
+theta = np.random.randn(2,1)
+
+for epoch in range(n_epochs):
+# Can you figure out a better way of setting up the contributions to each batch?
+ for i in range(m):
+ random_index = np.random.randint(m)
+ xi = X[random_index*M:random_index*M+M]
+ yi = y[random_index*M:random_index*M+M]
+ gradients = (2.0/M)* xi.T @ ((xi @ theta)-yi)
+ eta = learning_schedule(epoch*m+i)
+ theta = theta - eta*gradients
+print("theta from own sdg")
+print(theta)
+
+plt.plot(xnew, ypredict, "r-")
+plt.plot(xnew, ypredict2, "b-")
+plt.plot(x, y ,'ro')
+plt.axis([0,2.0,0, 15.0])
+plt.xlabel(r'$x$')
+plt.ylabel(r'$y$')
+plt.title(r'Random numbers ')
+plt.show()