From b6da39ee8204d0a8a4f05cf8855ad481d7ffa095 Mon Sep 17 00:00:00 2001 From: mhjensen Date: Thu, 19 Sep 2019 15:10:39 +0200 Subject: [PATCH] Updated gradient slides --- doc/pub/Splines/html/._Splines-bs000.html | 323 +- doc/pub/Splines/html/._Splines-bs001.html | 331 +- doc/pub/Splines/html/._Splines-bs002.html | 344 +- doc/pub/Splines/html/._Splines-bs003.html | 347 +- doc/pub/Splines/html/._Splines-bs004.html | 347 +- doc/pub/Splines/html/._Splines-bs005.html | 346 +- doc/pub/Splines/html/._Splines-bs006.html | 356 +- doc/pub/Splines/html/._Splines-bs007.html | 361 +- doc/pub/Splines/html/._Splines-bs008.html | 380 +- doc/pub/Splines/html/._Splines-bs009.html | 376 +- doc/pub/Splines/html/._Splines-bs010.html | 348 +- doc/pub/Splines/html/._Splines-bs011.html | 348 +- doc/pub/Splines/html/._Splines-bs012.html | 348 +- doc/pub/Splines/html/._Splines-bs013.html | 342 +- doc/pub/Splines/html/._Splines-bs014.html | 337 +- doc/pub/Splines/html/._Splines-bs015.html | 359 +- doc/pub/Splines/html/._Splines-bs016.html | 362 +- doc/pub/Splines/html/._Splines-bs017.html | 359 +- doc/pub/Splines/html/._Splines-bs018.html | 362 +- doc/pub/Splines/html/._Splines-bs019.html | 353 +- doc/pub/Splines/html/._Splines-bs020.html | 345 +- doc/pub/Splines/html/._Splines-bs021.html | 351 +- doc/pub/Splines/html/._Splines-bs022.html | 363 +- doc/pub/Splines/html/._Splines-bs023.html | 357 +- doc/pub/Splines/html/._Splines-bs024.html | 395 +- doc/pub/Splines/html/._Splines-bs025.html | 368 +- doc/pub/Splines/html/._Splines-bs026.html | 415 +-- doc/pub/Splines/html/._Splines-bs027.html | 356 +- doc/pub/Splines/html/._Splines-bs028.html | 344 +- doc/pub/Splines/html/._Splines-bs029.html | 357 +- doc/pub/Splines/html/._Splines-bs030.html | 357 +- doc/pub/Splines/html/._Splines-bs031.html | 372 +- doc/pub/Splines/html/._Splines-bs032.html | 357 +- doc/pub/Splines/html/._Splines-bs033.html | 385 +- doc/pub/Splines/html/._Splines-bs034.html | 419 ++- doc/pub/Splines/html/._Splines-bs035.html | 364 +- doc/pub/Splines/html/._Splines-bs036.html | 354 +- doc/pub/Splines/html/._Splines-bs037.html | 348 +- doc/pub/Splines/html/._Splines-bs038.html | 354 +- doc/pub/Splines/html/._Splines-bs039.html | 345 +- doc/pub/Splines/html/._Splines-bs040.html | 337 +- doc/pub/Splines/html/._Splines-bs041.html | 366 +- doc/pub/Splines/html/._Splines-bs042.html | 370 +- doc/pub/Splines/html/._Splines-bs043.html | 378 +- doc/pub/Splines/html/._Splines-bs044.html | 379 +- doc/pub/Splines/html/._Splines-bs045.html | 384 +- doc/pub/Splines/html/._Splines-bs046.html | 358 +- doc/pub/Splines/html/._Splines-bs047.html | 367 +- doc/pub/Splines/html/._Splines-bs048.html | 350 +- doc/pub/Splines/html/._Splines-bs049.html | 363 +- doc/pub/Splines/html/._Splines-bs050.html | 354 +- doc/pub/Splines/html/._Splines-bs051.html | 361 +- doc/pub/Splines/html/._Splines-bs052.html | 371 +- doc/pub/Splines/html/._Splines-bs053.html | 344 +- doc/pub/Splines/html/._Splines-bs054.html | 377 +- doc/pub/Splines/html/._Splines-bs055.html | 341 +- doc/pub/Splines/html/._Splines-bs056.html | 342 +- doc/pub/Splines/html/._Splines-bs057.html | 354 +- doc/pub/Splines/html/._Splines-bs058.html | 373 +- doc/pub/Splines/html/._Splines-bs059.html | 335 +- doc/pub/Splines/html/._Splines-bs060.html | 374 +- doc/pub/Splines/html/._Splines-bs061.html | 363 +- doc/pub/Splines/html/._Splines-bs062.html | 412 ++- doc/pub/Splines/html/._Splines-bs063.html | 414 +-- doc/pub/Splines/html/._Splines-bs064.html | 342 +- doc/pub/Splines/html/._Splines-bs065.html | 353 +- doc/pub/Splines/html/._Splines-bs066.html | 349 +- doc/pub/Splines/html/._Splines-bs067.html | 363 +- doc/pub/Splines/html/._Splines-bs068.html | 362 +- doc/pub/Splines/html/._Splines-bs069.html | 352 +- doc/pub/Splines/html/._Splines-bs070.html | 365 +- doc/pub/Splines/html/._Splines-bs071.html | 362 +- doc/pub/Splines/html/._Splines-bs072.html | 356 +- doc/pub/Splines/html/Splines-bs.html | 323 +- doc/pub/Splines/html/Splines-reveal.html | 2205 +++++------ doc/pub/Splines/html/Splines-solarized.html | 2265 ++++++------ doc/pub/Splines/html/Splines.html | 2265 ++++++------ doc/pub/Splines/html/reveal.js/.gitignore | 7 +- doc/pub/Splines/html/reveal.js/.travis.yml | 6 +- doc/pub/Splines/html/reveal.js/LICENSE | 2 +- doc/pub/Splines/html/reveal.js/README.md | 592 +-- doc/pub/Splines/html/reveal.js/bower.json | 6 +- .../html/reveal.js/css/print/paper.css | 7 +- .../Splines/html/reveal.js/css/print/pdf.css | 97 +- .../Splines/html/reveal.js/css/reveal.scss | 588 +-- .../html/reveal.js/css/theme/README.md | 6 +- .../reveal.js/css/theme/source/black.scss | 4 +- .../reveal.js/css/theme/source/white.scss | 4 +- doc/pub/Splines/html/reveal.js/index.html | 390 +- doc/pub/Splines/html/reveal.js/js/reveal.js | 1419 ++----- .../html/reveal.js/lib/css/zenburn.css | 119 +- .../Splines/html/reveal.js/lib/js/head.min.js | 17 +- doc/pub/Splines/html/reveal.js/package.json | 42 +- .../reveal.js/plugin/highlight/highlight.js | 55 +- .../reveal.js/plugin/markdown/example.html | 7 - .../html/reveal.js/plugin/markdown/example.md | 5 - .../reveal.js/plugin/markdown/markdown.js | 57 +- .../html/reveal.js/plugin/markdown/marked.js | 2 +- .../html/reveal.js/plugin/math/math.js | 7 +- .../html/reveal.js/plugin/multiplex/client.js | 2 +- .../html/reveal.js/plugin/multiplex/index.js | 38 +- .../html/reveal.js/plugin/multiplex/master.js | 61 +- .../reveal.js/plugin/notes-server/client.js | 7 +- .../reveal.js/plugin/notes-server/index.js | 31 +- .../reveal.js/plugin/notes-server/notes.html | 241 +- .../html/reveal.js/plugin/notes/notes.html | 417 +-- .../html/reveal.js/plugin/notes/notes.js | 45 +- .../reveal.js/plugin/print-pdf/print-pdf.js | 75 +- .../html/reveal.js/plugin/search/search.js | 72 +- .../html/reveal.js/plugin/zoom-js/zoom.js | 40 +- .../html/reveal.js/test/examples/math.html | 2 +- .../test/examples/slide-backgrounds.html | 2 +- .../html/reveal.js/test/test-markdown.html | 2 +- doc/pub/Splines/html/reveal.js/test/test.html | 3 +- doc/pub/Splines/html/reveal.js/test/test.js | 10 +- doc/pub/Splines/ipynb/Splines.ipynb | 3294 ++++++++--------- .../Splines/ipynb/ipynb-Splines-src.tar.gz | Bin 209 -> 211 bytes doc/pub/Splines/pdf/Splines-minted.pdf | Bin 452147 -> 357363 bytes doc/src/Splines/Splines.do.txt | 1991 +++++----- doc/src/Splines/Splines.do.txt~ | 1864 ---------- 120 files changed, 20811 insertions(+), 24189 deletions(-) delete mode 100644 doc/src/Splines/Splines.do.txt~ diff --git a/doc/pub/Splines/html/._Splines-bs000.html b/doc/pub/Splines/html/._Splines-bs000.html index 9941cf5dd..b905e18e2 100644 --- a/doc/pub/Splines/html/._Splines-bs000.html +++ b/doc/pub/Splines/html/._Splines-bs000.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -289,7 +292,7 @@ MathJax.Hub.Config({
[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

-

Oct 18, 2018

+

Sep 19, 2019


@@ -313,7 +316,7 @@ MathJax.Hub.Config({

  • 9
  • 10
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • @@ -331,7 +334,7 @@ MathJax.Hub.Config({
    - © 1999-2018, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license + © 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
    diff --git a/doc/pub/Splines/html/._Splines-bs001.html b/doc/pub/Splines/html/._Splines-bs001.html index 7c96aed68..6ed46cf6a 100644 --- a/doc/pub/Splines/html/._Splines-bs001.html +++ b/doc/pub/Splines/html/._Splines-bs001.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,17 +273,7 @@ MathJax.Hub.Config({ -

    Optimization, the central part of any Machine Learning algortithm

    - -

    -Almost every problem in machine learning and data science starts with -a dataset \( X \), a model \( g(\beta) \), which is a function of the -parameters \( \beta \) and a cost function \( C(X, g(\beta)) \) that allows -us to judge how well the model \( g(\beta) \) explains the observations -\( X \). The model is fit by finding the values of \( \beta \) that minimize -the cost function. Ideally we would be able to solve for \( \beta \) -analytically, however this is not possible in general and we must use -some approximative/numerical method to compute the minimum. +

    Optimization problems, why?

    @@ -299,7 +292,7 @@ some approximative/numerical method to compute the minimum.

  • 10
  • 11
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs002.html b/doc/pub/Splines/html/._Splines-bs002.html index e2973ea7b..0074b4339 100644 --- a/doc/pub/Splines/html/._Splines-bs002.html +++ b/doc/pub/Splines/html/._Splines-bs002.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,24 +273,17 @@ MathJax.Hub.Config({ -

    Revisiting our Logistic Regression case

    +

    Optimization, the central part of any Machine Learning algortithm

    -In our discussion on Logistic Regression we studied the -case of -two classes, with \( y_i \) either -\( 0 \) or \( 1 \). Furthermore we assumed also that we have only two -parameters \( \beta \) in our fitting, that is we -defined probabilities - -$$ -\begin{align*} -p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\ -p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}), -\end{align*} -$$ - -where \( \hat{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \). +Almost every problem in machine learning and data science starts with +a dataset \( X \), a model \( g(\beta) \), which is a function of the +parameters \( \beta \) and a cost function \( C(X, g(\beta)) \) that allows +us to judge how well the model \( g(\beta) \) explains the observations +\( X \). The model is fit by finding the values of \( \beta \) that minimize +the cost function. Ideally we would be able to solve for \( \beta \) +analytically, however this is not possible in general and we must use +some approximative/numerical method to compute the minimum.

    @@ -307,7 +303,7 @@ where \( \hat{\beta} \) are the weights we wish to extract from data, in our cas

  • 11
  • 12
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs003.html b/doc/pub/Splines/html/._Splines-bs003.html index 92f52902c..363a6da93 100644 --- a/doc/pub/Splines/html/._Splines-bs003.html +++ b/doc/pub/Splines/html/._Splines-bs003.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,28 +273,24 @@ MathJax.Hub.Config({ -

    The equations to solve

    +

    Revisiting our Logistic Regression case

    -Our compact equations used a definition of a vector \( \hat{y} \) with \( n \) -elements \( y_i \), an \( n\times p \) matrix \( \hat{X} \) which contains the -\( x_i \) values and a vector \( \hat{p} \) of fitted probabilities -\( p(y_i\vert x_i,\hat{\beta}) \). We rewrote in a more compact form -the first derivative of the cost function as +In our discussion on Logistic Regression we studied the +case of +two classes, with \( y_i \) either +\( 0 \) or \( 1 \). Furthermore we assumed also that we have only two +parameters \( \beta \) in our fitting, that is we +defined probabilities $$ -\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right). +\begin{align*} +p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\ +p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}), +\end{align*} $$ -

    -If we in addition define a diagonal matrix \( \hat{W} \) with elements -\( p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta}) \), we can obtain a compact expression of the second derivative as - -$$ -\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}. -$$ - -This defines what is called the Hessian matrix. +where \( \hat{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).

    @@ -312,7 +311,7 @@ This defines what is called the Hessian matrix.

  • 12
  • 13
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs004.html b/doc/pub/Splines/html/._Splines-bs004.html index 8ab5c2d11..3436d9ff1 100644 --- a/doc/pub/Splines/html/._Splines-bs004.html +++ b/doc/pub/Splines/html/._Splines-bs004.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,28 +273,28 @@ MathJax.Hub.Config({ -

    Solving using Newton-Raphson's method

    +

    The equations to solve

    -If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives. +Our compact equations used a definition of a vector \( \hat{y} \) with \( n \) +elements \( y_i \), an \( n\times p \) matrix \( \hat{X} \) which contains the +\( x_i \) values and a vector \( \hat{p} \) of fitted probabilities +\( p(y_i\vert x_i,\hat{\beta}) \). We rewrote in a more compact form +the first derivative of the cost function as + +$$ +\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right). +$$

    -Our iterative scheme is then given by +If we in addition define a diagonal matrix \( \hat{W} \) with elements +\( p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta}) \), we can obtain a compact expression of the second derivative as $$ -\hat{\beta}^{\mathrm{new}} = \hat{\beta}^{\mathrm{old}}-\left(\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T}\right)^{-1}_{\hat{\beta}^{\mathrm{old}}}\times \left(\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}}\right)_{\hat{\beta}^{\mathrm{old}}}, +\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}. $$ -or in matrix form as - -$$ -\hat{\beta}^{\mathrm{new}} = \hat{\beta}^{\mathrm{old}}-\left(\hat{X}^T\hat{W}\hat{X} \right)^{-1}\times \left(-\hat{X}^T(\hat{y}-\hat{p}) \right)_{\hat{\beta}^{\mathrm{old}}}. -$$ - -The right-hand side is computed with the old values of \( \beta \). - -

    -If we can compute these matrices, in particular the Hessian, the above is often the easiest method to implement. +This defines what is called the Hessian matrix.

    @@ -313,7 +316,7 @@ If we can compute these matrices, in particular the Hessian, the above is often

  • 13
  • 14
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs005.html b/doc/pub/Splines/html/._Splines-bs005.html index a02b6e8bd..d9750a3a0 100644 --- a/doc/pub/Splines/html/._Splines-bs005.html +++ b/doc/pub/Splines/html/._Splines-bs005.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,19 +273,28 @@ MathJax.Hub.Config({ -

    Brief reminder on Newton-Raphson's method

    +

    Solving using Newton-Raphson's method

    -Let us quickly remind ourselves how we derive the above method. +If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives.

    -Perhaps the most celebrated of all one-dimensional root-finding -routines is Newton's method, also called the Newton-Raphson -method. This method requires the evaluation of both the -function \( f \) and its derivative \( f' \) at arbitrary points. -If you can only calculate the derivative -numerically and/or your function is not of the smooth type, we -normally discourage the use of this method. +Our iterative scheme is then given by + +$$ +\hat{\beta}^{\mathrm{new}} = \hat{\beta}^{\mathrm{old}}-\left(\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T}\right)^{-1}_{\hat{\beta}^{\mathrm{old}}}\times \left(\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}}\right)_{\hat{\beta}^{\mathrm{old}}}, +$$ + +or in matrix form as + +$$ +\hat{\beta}^{\mathrm{new}} = \hat{\beta}^{\mathrm{old}}-\left(\hat{X}^T\hat{W}\hat{X} \right)^{-1}\times \left(-\hat{X}^T(\hat{y}-\hat{p}) \right)_{\hat{\beta}^{\mathrm{old}}}. +$$ + +The right-hand side is computed with the old values of \( \beta \). + +

    +If we can compute these matrices, in particular the Hessian, the above is often the easiest method to implement.

    @@ -305,7 +317,7 @@ normally discourage the use of this method.

  • 14
  • 15
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs006.html b/doc/pub/Splines/html/._Splines-bs006.html index c54b74606..ba510b274 100644 --- a/doc/pub/Splines/html/._Splines-bs006.html +++ b/doc/pub/Splines/html/._Splines-bs006.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,38 +273,19 @@ MathJax.Hub.Config({ -

    The equations

    +

    Brief reminder on Newton-Raphson's method

    -The Newton-Raphson formula consists geometrically of extending the -tangent line at a current point until it crosses zero, then setting -the next guess to the abscissa of that zero-crossing. The mathematics -behind this method is rather simple. Employing a Taylor expansion for -\( x \) sufficiently close to the solution \( s \), we have - -$$ - f(s)=0=f(x)+(s-x)f'(x)+\frac{(s-x)^2}{2}f''(x) +\dots. - \tag{1} -$$ +Let us quickly remind ourselves how we derive the above method.

    -For small enough values of the function and for well-behaved -functions, the terms beyond linear are unimportant, hence we obtain - -$$ - f(x)+(s-x)f'(x)\approx 0, -$$ - -yielding -$$ - s\approx x-\frac{f(x)}{f'(x)}. -$$ - -

    -Having in mind an iterative procedure, it is natural to start iterating with -$$ - x_{n+1}=x_n-\frac{f(x_n)}{f'(x_n)}. -$$ +Perhaps the most celebrated of all one-dimensional root-finding +routines is Newton's method, also called the Newton-Raphson +method. This method requires the evaluation of both the +function \( f \) and its derivative \( f' \) at arbitrary points. +If you can only calculate the derivative +numerically and/or your function is not of the smooth type, we +normally discourage the use of this method.

    @@ -325,7 +309,7 @@ $$

  • 15
  • 16
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs007.html b/doc/pub/Splines/html/._Splines-bs007.html index 5d470e6d7..62f7f41a4 100644 --- a/doc/pub/Splines/html/._Splines-bs007.html +++ b/doc/pub/Splines/html/._Splines-bs007.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,20 +273,38 @@ MathJax.Hub.Config({ -

    Simple geometric interpretation

    +

    The equations

    -The above is Newton-Raphson's method. It has a simple geometric -interpretation, namely \( x_{n+1} \) is the point where the tangent from -\( (x_n,f(x_n)) \) crosses the \( x \)-axis. Close to the solution, -Newton-Raphson converges fast to the desired result. However, if we -are far from a root, where the higher-order terms in the series are -important, the Newton-Raphson formula can give grossly inaccurate -results. For instance, the initial guess for the root might be so far -from the true root as to let the search interval include a local -maximum or minimum of the function. If an iteration places a trial -guess near such a local extremum, so that the first derivative nearly -vanishes, then Newton-Raphson may fail totally +The Newton-Raphson formula consists geometrically of extending the +tangent line at a current point until it crosses zero, then setting +the next guess to the abscissa of that zero-crossing. The mathematics +behind this method is rather simple. Employing a Taylor expansion for +\( x \) sufficiently close to the solution \( s \), we have + +$$ + f(s)=0=f(x)+(s-x)f'(x)+\frac{(s-x)^2}{2}f''(x) +\dots. + \tag{1} +$$ + +

    +For small enough values of the function and for well-behaved +functions, the terms beyond linear are unimportant, hence we obtain + +$$ + f(x)+(s-x)f'(x)\approx 0, +$$ + +yielding +$$ + s\approx x-\frac{f(x)}{f'(x)}. +$$ + +

    +Having in mind an iterative procedure, it is natural to start iterating with +$$ + x_{n+1}=x_n-\frac{f(x_n)}{f'(x_n)}. +$$

    @@ -308,7 +329,7 @@ vanishes, then Newton-Raphson may fail totally

  • 16
  • 17
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs008.html b/doc/pub/Splines/html/._Splines-bs008.html index 4ad5559bc..556279ca3 100644 --- a/doc/pub/Splines/html/._Splines-bs008.html +++ b/doc/pub/Splines/html/._Splines-bs008.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,57 +273,20 @@ MathJax.Hub.Config({ -

    Extending to more than one variable

    +

    Simple geometric interpretation

    -Newton's method can be generalized to systems of several non-linear equations -and variables. Consider the case with two equations -$$ - \begin{array}{cc} f_1(x_1,x_2) &=0\\ - f_2(x_1,x_2) &=0,\end{array} -$$ - -which we Taylor expand to obtain - -$$ - \begin{array}{cc} 0=f_1(x_1+h_1,x_2+h_2)=&f_1(x_1,x_2)+h_1 - \partial f_1/\partial x_1+h_2 - \partial f_1/\partial x_2+\dots\\ - 0=f_2(x_1+h_1,x_2+h_2)=&f_2(x_1,x_2)+h_1 - \partial f_2/\partial x_1+h_2 - \partial f_2/\partial x_2+\dots - \end{array}. -$$ - -Defining the Jacobian matrix \( {\bf \hat{J}} \) we have -$$ - {\bf \hat{J}}=\left( \begin{array}{cc} - \partial f_1/\partial x_1 & \partial f_1/\partial x_2 \\ - \partial f_2/\partial x_1 &\partial f_2/\partial x_2 - \end{array} \right), -$$ - -we can rephrase Newton's method as -$$ -\left(\begin{array}{c} x_1^{n+1} \\ x_2^{n+1} \end{array} \right)= -\left(\begin{array}{c} x_1^{n} \\ x_2^{n} \end{array} \right)+ -\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right), -$$ - -where we have defined -$$ - \left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right)= - -{\bf \hat{J}}^{-1} - \left(\begin{array}{c} f_1(x_1^{n},x_2^{n}) \\ f_2(x_1^{n},x_2^{n}) \end{array} \right). -$$ - -We need thus to compute the inverse of the Jacobian matrix and it -is to understand that difficulties may -arise in case \( {\bf \hat{J}} \) is nearly singular. - -

    -It is rather straightforward to extend the above scheme to systems of -more than two non-linear equations. In our case, the Jacobian matrix is given by the Hessian that represents the second derivative of cost function. +The above is Newton-Raphson's method. It has a simple geometric +interpretation, namely \( x_{n+1} \) is the point where the tangent from +\( (x_n,f(x_n)) \) crosses the \( x \)-axis. Close to the solution, +Newton-Raphson converges fast to the desired result. However, if we +are far from a root, where the higher-order terms in the series are +important, the Newton-Raphson formula can give grossly inaccurate +results. For instance, the initial guess for the root might be so far +from the true root as to let the search interval include a local +maximum or minimum of the function. If an iteration places a trial +guess near such a local extremum, so that the first derivative nearly +vanishes, then Newton-Raphson may fail totally

    @@ -346,7 +312,7 @@ more than two non-linear equations. In our case, the Jacobian matrix is given by

  • 17
  • 18
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs009.html b/doc/pub/Splines/html/._Splines-bs009.html index affc4df41..2ca3663a4 100644 --- a/doc/pub/Splines/html/._Splines-bs009.html +++ b/doc/pub/Splines/html/._Splines-bs009.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,26 +273,57 @@ MathJax.Hub.Config({ -

    Steepest descent

    +

    Extending to more than one variable

    -The basic idea of gradient descent is -that a function \( F(\mathbf{x}) \), -\( \mathbf{x} \equiv (x_1,\cdots,x_n) \), decreases fastest if one goes from \( \bf {x} \) in the -direction of the negative gradient \( -\nabla F(\mathbf{x}) \). - -

    -It can be shown that if +Newton's method can be generalized to systems of several non-linear equations +and variables. Consider the case with two equations $$ -\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), + \begin{array}{cc} f_1(x_1,x_2) &=0\\ + f_2(x_1,x_2) &=0,\end{array} $$ -with \( \gamma_k > 0 \). +which we Taylor expand to obtain + +$$ + \begin{array}{cc} 0=f_1(x_1+h_1,x_2+h_2)=&f_1(x_1,x_2)+h_1 + \partial f_1/\partial x_1+h_2 + \partial f_1/\partial x_2+\dots\\ + 0=f_2(x_1+h_1,x_2+h_2)=&f_2(x_1,x_2)+h_1 + \partial f_2/\partial x_1+h_2 + \partial f_2/\partial x_2+\dots + \end{array}. +$$ + +Defining the Jacobian matrix \( {\bf \hat{J}} \) we have +$$ + {\bf \hat{J}}=\left( \begin{array}{cc} + \partial f_1/\partial x_1 & \partial f_1/\partial x_2 \\ + \partial f_2/\partial x_1 &\partial f_2/\partial x_2 + \end{array} \right), +$$ + +we can rephrase Newton's method as +$$ +\left(\begin{array}{c} x_1^{n+1} \\ x_2^{n+1} \end{array} \right)= +\left(\begin{array}{c} x_1^{n} \\ x_2^{n} \end{array} \right)+ +\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right), +$$ + +where we have defined +$$ + \left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right)= + -{\bf \hat{J}}^{-1} + \left(\begin{array}{c} f_1(x_1^{n},x_2^{n}) \\ f_2(x_1^{n},x_2^{n}) \end{array} \right). +$$ + +We need thus to compute the inverse of the Jacobian matrix and it +is to understand that difficulties may +arise in case \( {\bf \hat{J}} \) is nearly singular.

    -For \( \gamma_k \) small enough, then \( F(\mathbf{x}_{k+1}) \leq -F(\mathbf{x}_k) \). This means that for a sufficiently small \( \gamma_k \) -we are always moving towards smaller function values, i.e a minimum. +It is rather straightforward to extend the above scheme to systems of +more than two non-linear equations. In our case, the Jacobian matrix is given by the Hessian that represents the second derivative of cost function.

    @@ -316,7 +350,7 @@ we are always moving towards smaller function values, i.e a minimum.

  • 18
  • 19
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs010.html b/doc/pub/Splines/html/._Splines-bs010.html index c648a314c..8884c647d 100644 --- a/doc/pub/Splines/html/._Splines-bs010.html +++ b/doc/pub/Splines/html/._Splines-bs010.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,23 +271,28 @@ MathJax.Hub.Config({

     

     

     

    - + -

    More on Steepest descent

    +

    Steepest descent

    -The previous observation is the basis of the method of steepest -descent, which is also referred to as just gradient descent (GD). One -starts with an initial guess \( \mathbf{x}_0 \) for a minimum of \( F \) and -computes new approximations according to - -$$ -\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), \ \ k \geq 0. -$$ +The basic idea of gradient descent is +that a function \( F(\mathbf{x}) \), +\( \mathbf{x} \equiv (x_1,\cdots,x_n) \), decreases fastest if one goes from \( \bf {x} \) in the +direction of the negative gradient \( -\nabla F(\mathbf{x}) \).

    -The parameter \( \gamma_k \) is often referred to as the step length or -the learning rate within the context of Machine Learning. +It can be shown that if +$$ +\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), +$$ + +with \( \gamma_k > 0 \). + +

    +For \( \gamma_k \) small enough, then \( F(\mathbf{x}_{k+1}) \leq +F(\mathbf{x}_k) \). This means that for a sufficiently small \( \gamma_k \) +we are always moving towards smaller function values, i.e a minimum.

    @@ -312,7 +320,7 @@ the learning rate within the context of Machine Learning.

  • 19
  • 20
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs011.html b/doc/pub/Splines/html/._Splines-bs011.html index 5906abe1a..364955e0c 100644 --- a/doc/pub/Splines/html/._Splines-bs011.html +++ b/doc/pub/Splines/html/._Splines-bs011.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,28 +273,21 @@ MathJax.Hub.Config({ -

    The ideal

    +

    More on Steepest descent

    -Ideally the sequence \( \{\mathbf{x}_k \}_{k=0} \) converges to a global -minimum of the function \( F \). In general we do not know if we are in a -global or local minimum. In the special case when \( F \) is a convex -function, all local minima are also global minima, so in this case -gradient descent can converge to the global solution. The advantage of -this scheme is that it is conceptually simple and straightforward to -implement. However the method in this form has some severe -limitations: +The previous observation is the basis of the method of steepest +descent, which is also referred to as just gradient descent (GD). One +starts with an initial guess \( \mathbf{x}_0 \) for a minimum of \( F \) and +computes new approximations according to + +$$ +\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), \ \ k \geq 0. +$$

    -In machine learing we are often faced with non-convex high dimensional -cost functions with many local minima. Since GD is deterministic we -will get stuck in a local minimum, if the method converges, unless we -have a very good intial guess. This also implies that the scheme is -sensitive to the chosen initial condition. - -

    -Note that the gradient is a function of \( \mathbf{x} = -(x_1,\cdots,x_n) \) which makes it expensive to compute numerically. +The parameter \( \gamma_k \) is often referred to as the step length or +the learning rate within the context of Machine Learning.

    @@ -319,7 +315,7 @@ Note that the gradient is a function of \( \mathbf{x} =

  • 20
  • 21
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs012.html b/doc/pub/Splines/html/._Splines-bs012.html index 46aa7b80f..1b2f5ee12 100644 --- a/doc/pub/Splines/html/._Splines-bs012.html +++ b/doc/pub/Splines/html/._Splines-bs012.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,21 +273,28 @@ MathJax.Hub.Config({ -

    The sensitiveness of the gradient descent

    +

    The ideal

    -The gradient descent method -is sensitive to the choice of learning rate \( \gamma_k \). This is due -to the fact that we are only guaranteed that \( F(\mathbf{x}_{k+1}) \leq -F(\mathbf{x}_k) \) for sufficiently small \( \gamma_k \). The problem is to -determine an optimal learning rate. If the learning rate is chosen too -small the method will take a long time to converge and if it is too -large we can experience erratic behavior. +Ideally the sequence \( \{\mathbf{x}_k \}_{k=0} \) converges to a global +minimum of the function \( F \). In general we do not know if we are in a +global or local minimum. In the special case when \( F \) is a convex +function, all local minima are also global minima, so in this case +gradient descent can converge to the global solution. The advantage of +this scheme is that it is conceptually simple and straightforward to +implement. However the method in this form has some severe +limitations:

    -Many of these shortcomings can be alleviated by introducing -randomness. One such method is that of Stochastic Gradient Descent -(SGD), see below. +In machine learing we are often faced with non-convex high dimensional +cost functions with many local minima. Since GD is deterministic we +will get stuck in a local minimum, if the method converges, unless we +have a very good intial guess. This also implies that the scheme is +sensitive to the chosen initial condition. + +

    +Note that the gradient is a function of \( \mathbf{x} = +(x_1,\cdots,x_n) \) which makes it expensive to compute numerically.

    @@ -312,7 +322,7 @@ randomness. One such method is that of Stochastic Gradient Descent

  • 21
  • 22
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs013.html b/doc/pub/Splines/html/._Splines-bs013.html index 024257641..05f78d0e3 100644 --- a/doc/pub/Splines/html/._Splines-bs013.html +++ b/doc/pub/Splines/html/._Splines-bs013.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,22 +273,21 @@ MathJax.Hub.Config({ -

    Convex functions

    +

    The sensitiveness of the gradient descent

    -Ideally we want our cost/loss function to be convex(concave). +The gradient descent method +is sensitive to the choice of learning rate \( \gamma_k \). This is due +to the fact that we are only guaranteed that \( F(\mathbf{x}_{k+1}) \leq +F(\mathbf{x}_k) \) for sufficiently small \( \gamma_k \). The problem is to +determine an optimal learning rate. If the learning rate is chosen too +small the method will take a long time to converge and if it is too +large we can experience erratic behavior.

    -First we give the definition of a convex set: A set \( C \) in -\( \mathbb{R}^n \) is said to be convex if, for all \( x \) and \( y \) in \( C \) and -all \( t \in (0,1) \) , the point \( (1 − t)x + ty \) also belongs to -C. Geometrically this means that every point on the line segment -connecting \( x \) and \( y \) is in \( C \) as discussed below. - -

    -The convex subsets of \( \mathbb{R} \) are the intervals of -\( \mathbb{R} \). Examples of convex sets of \( \mathbb{R}^2 \) are the -regular polygons (triangles, rectangles, pentagons, etc...). +Many of these shortcomings can be alleviated by introducing +randomness. One such method is that of Stochastic Gradient Descent +(SGD), see below.

    @@ -313,7 +315,7 @@ regular polygons (triangles, rectangles, pentagons, etc...).

  • 22
  • 23
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs014.html b/doc/pub/Splines/html/._Splines-bs014.html index 791b919c0..9104381d4 100644 --- a/doc/pub/Splines/html/._Splines-bs014.html +++ b/doc/pub/Splines/html/._Splines-bs014.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,12 +271,24 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Convex function

    +

    Convex functions

    -Convex function: Let \( X \subset \mathbb{R}^n \) be a convex set. Assume that the function \( f: X \rightarrow \mathbb{R} \) is continuous, then \( f \) is said to be convex if $$f(tx_1 + (1-t)x_2) \leq tf(x_1) + (1-t)f(x_2) $$ for all \( x_1, x_2 \in X \) and for all \( t \in [0,1] \). If \( \leq \) is replaced with a strict inequaltiy in the definition, we demand \( x_1 \neq x_2 \) and \( t\in(0,1) \) then \( f \) is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting \( f(x_1) \) and \( f(x_2) \), the value of the function on the interval \( [x_1,x_2] \) is always below the line as illustrated below. +Ideally we want our cost/loss function to be convex(concave). + +

    +First we give the definition of a convex set: A set \( C \) in +\( \mathbb{R}^n \) is said to be convex if, for all \( x \) and \( y \) in \( C \) and +all \( t \in (0,1) \) , the point \( (1 − t)x + ty \) also belongs to +C. Geometrically this means that every point on the line segment +connecting \( x \) and \( y \) is in \( C \) as discussed below. + +

    +The convex subsets of \( \mathbb{R} \) are the intervals of +\( \mathbb{R} \). Examples of convex sets of \( \mathbb{R}^2 \) are the +regular polygons (triangles, rectangles, pentagons, etc...).

    @@ -301,7 +316,7 @@ MathJax.Hub.Config({

  • 23
  • 24
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs015.html b/doc/pub/Splines/html/._Splines-bs015.html index b26664033..f7e46720b 100644 --- a/doc/pub/Splines/html/._Splines-bs015.html +++ b/doc/pub/Splines/html/._Splines-bs015.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,46 +273,10 @@ MathJax.Hub.Config({ -

    Conditions on convex functions

    +

    Convex function

    -In the following we state first and second-order conditions which -ensures convexity of a function \( f \). We write \( D_f \) to denote the -domain of \( f \), i.e the subset of \( R^n \) where \( f \) is defined. For more -details and proofs we refer to: S. Boyd and L. Vandenberghe. Convex Optimization. Cambridge University Press. - -

    -

    -
    -

    -Suppose \( f \) is differentiable (i.e \( \nabla f(x) \) is well defined for -all \( x \) in the domain of \( f \)). Then \( f \) is convex if and only if \( D_f \) -is a convex set and $$f(y) \geq f(x) + \nabla f(x)^T (y-x) $$ holds -for all \( x,y \in D_f \). This condition means that for a convex function -the first order Taylor expansion (right hand side above) at any point -a global under estimator of the function. To convince yourself you can -make a drawing of \( f(x) = x^2+1 \) and draw the tangent line to \( f(x) \) and -note that it is always below the graph. -

    -
    - - -

    -

    -
    -

    -Assume that \( f \) is twice -differentiable, i.e the Hessian matrix exists at each point in -\( D_f \). Then \( f \) is convex if and only if \( D_f \) is a convex set and its -Hessian is positive semi-definite for all \( x\in D_f \). For a -single-variable function this reduces to \( f''(x) \geq 0 \). Geometrically this means that \( f \) has nonnegative curvature -everywhere. -

    -
    - - -

    -This condition is particularly useful since it gives us an procedure for determining if the function under consideration is convex, apart from using the definition. +Convex function: Let \( X \subset \mathbb{R}^n \) be a convex set. Assume that the function \( f: X \rightarrow \mathbb{R} \) is continuous, then \( f \) is said to be convex if $$f(tx_1 + (1-t)x_2) \leq tf(x_1) + (1-t)f(x_2) $$ for all \( x_1, x_2 \in X \) and for all \( t \in [0,1] \). If \( \leq \) is replaced with a strict inequaltiy in the definition, we demand \( x_1 \neq x_2 \) and \( t\in(0,1) \) then \( f \) is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting \( f(x_1) \) and \( f(x_2) \), the value of the function on the interval \( [x_1,x_2] \) is always below the line as illustrated below.

    @@ -337,7 +304,7 @@ This condition is particularly useful since it gives us an procedure for determi

  • 24
  • 25
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs016.html b/doc/pub/Splines/html/._Splines-bs016.html index d136e2748..2128670ed 100644 --- a/doc/pub/Splines/html/._Splines-bs016.html +++ b/doc/pub/Splines/html/._Splines-bs016.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,33 +273,46 @@ MathJax.Hub.Config({ -

    More on convex functions

    +

    Conditions on convex functions

    -The next result is of great importance to us and the reason why we are -going on about convex functions. In machine learning we frequently -have to minimize a loss/cost function in order to find the best -parameters for the model we are considering. - -

    -Ideally we want the -global minimum (for high-dimensional models it is hard to know -if we have local or global minimum). However, if the cost/loss function -is convex the following result provides invaluable information: +In the following we state first and second-order conditions which +ensures convexity of a function \( f \). We write \( D_f \) to denote the +domain of \( f \), i.e the subset of \( R^n \) where \( f \) is defined. For more +details and proofs we refer to: S. Boyd and L. Vandenberghe. Convex Optimization. Cambridge University Press.

    -Consider the problem of finding \( x \in \mathbb{R}^n \) such that \( f(x) \) -is minimal, where \( f \) is convex and differentiable. Then, any point -\( x^* \) that satisfies \( \nabla f(x^*) = 0 \) is a global minimum. +Suppose \( f \) is differentiable (i.e \( \nabla f(x) \) is well defined for +all \( x \) in the domain of \( f \)). Then \( f \) is convex if and only if \( D_f \) +is a convex set and $$f(y) \geq f(x) + \nabla f(x)^T (y-x) $$ holds +for all \( x,y \in D_f \). This condition means that for a convex function +the first order Taylor expansion (right hand side above) at any point +a global under estimator of the function. To convince yourself you can +make a drawing of \( f(x) = x^2+1 \) and draw the tangent line to \( f(x) \) and +note that it is always below the graph.

    -This result means that if we know that the cost/loss function is convex and we are able to find a minimum, we are guaranteed that it is a global minimum. +

    +
    +

    +Assume that \( f \) is twice +differentiable, i.e the Hessian matrix exists at each point in +\( D_f \). Then \( f \) is convex if and only if \( D_f \) is a convex set and its +Hessian is positive semi-definite for all \( x\in D_f \). For a +single-variable function this reduces to \( f''(x) \geq 0 \). Geometrically this means that \( f \) has nonnegative curvature +everywhere. +

    +
    + + +

    +This condition is particularly useful since it gives us an procedure for determining if the function under consideration is convex, apart from using the definition.

    @@ -324,7 +340,7 @@ This result means that if we know that the cost/loss function is convex and we a

  • 25
  • 26
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs017.html b/doc/pub/Splines/html/._Splines-bs017.html index f77f5e114..3979aa516 100644 --- a/doc/pub/Splines/html/._Splines-bs017.html +++ b/doc/pub/Splines/html/._Splines-bs017.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,29 +273,33 @@ MathJax.Hub.Config({ -

    Some simple problems

    +

    More on convex functions

    -
      -
    1. Show that \( f(x)=x^2 \) is convex for \( x \in \mathbb{R} \) using the definition of convexity. Hint: If you re-write the definition, \( f \) is convex if the following holds for all \( x,y \in D_f \) and any \( \lambda \in [0,1] \) $\lambda f(x)+(1-\lambda)f(y)-f(\lambda x + (1-\lambda) y ) \geq 0$.
    2. -
    3. Using the second order condition show that the following functions are convex on the specified domain.
    4. +

      +The next result is of great importance to us and the reason why we are +going on about convex functions. In machine learning we frequently +have to minimize a loss/cost function in order to find the best +parameters for the model we are considering. -

      +

      +Ideally we want the +global minimum (for high-dimensional models it is hard to know +if we have local or global minimum). However, if the cost/loss function +is convex the following result provides invaluable information: -

    5. Let \( f(x) = x^2 \) and \( g(x) = e^x \). Show that \( f(g(x)) \) and \( g(f(x)) \) is convex for \( x \in \mathbb{R} \). Also show that if \( f(x) \) is any convex function than \( h(x) = e^{f(x)} \) is convex.
    6. -
    7. A norm is any function that satisfy the following properties
    8. +

      +

      +
      +

      +Consider the problem of finding \( x \in \mathbb{R}^n \) such that \( f(x) \) +is minimal, where \( f \) is convex and differentiable. Then, any point +\( x^* \) that satisfies \( \nabla f(x^*) = 0 \) is a global minimum. +

      +
      - -
    - -Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this). +

    +This result means that if we know that the cost/loss function is convex and we are able to find a minimum, we are guaranteed that it is a global minimum.

    @@ -320,7 +327,7 @@ Using the definition of convexity, try to show that a function satisfying the pr

  • 26
  • 27
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs018.html b/doc/pub/Splines/html/._Splines-bs018.html index 843b2391f..0291e2424 100644 --- a/doc/pub/Splines/html/._Splines-bs018.html +++ b/doc/pub/Splines/html/._Splines-bs018.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,38 +273,29 @@ MathJax.Hub.Config({ -

    Standard steepest descent

    +

    Some simple problems

    -

    -Before we proceed, we would like to discuss the approach called the -standard Steepest descent, which again leads to us having to be able -to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG). +

      +
    1. Show that \( f(x)=x^2 \) is convex for \( x \in \mathbb{R} \) using the definition of convexity. Hint: If you re-write the definition, \( f \) is convex if the following holds for all \( x,y \in D_f \) and any \( \lambda \in [0,1] \) $\lambda f(x)+(1-\lambda)f(y)-f(\lambda x + (1-\lambda) y ) \geq 0$.
    2. +
    3. Using the second order condition show that the following functions are convex on the specified domain.
    4. -

      -The success of the CG method -for finding solutions of non-linear problems is based on the theory -of conjugate gradients for linear systems of equations. It belongs to -the class of iterative methods for solving problems from linear -algebra of the type -$$ -\begin{equation*} -\hat{A}\hat{x} = \hat{b}. -\end{equation*} -$$ +

      -

      -In the iterative process we end up with a problem like +

    5. Let \( f(x) = x^2 \) and \( g(x) = e^x \). Show that \( f(g(x)) \) and \( g(f(x)) \) is convex for \( x \in \mathbb{R} \). Also show that if \( f(x) \) is any convex function than \( h(x) = e^{f(x)} \) is convex.
    6. +
    7. A norm is any function that satisfy the following properties
    8. -$$ -\begin{equation*} - \hat{r}= \hat{b}-\hat{A}\hat{x}, -\end{equation*} -$$ + -where \( \hat{r} \) is the so-called residual or error in the iterative process. +
    -

    -When we have found the exact solution, \( \hat{r}=0 \). +Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this).

    @@ -329,7 +323,7 @@ When we have found the exact solution, \( \hat{r}=0 \).

  • 27
  • 28
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs019.html b/doc/pub/Splines/html/._Splines-bs019.html index 4744e84cf..fc2b4bc2c 100644 --- a/doc/pub/Splines/html/._Splines-bs019.html +++ b/doc/pub/Splines/html/._Splines-bs019.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,21 +271,37 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Gradient method

    +

    Revisiting our first homework

    -The residual is zero when we reach the minimum of the quadratic equation +We will use linear regression as a case study for the gradient descent +methods. Linear regression is a great test case for the gradient +descent methods discussed in the lectures since it has several +desirable properties such as: + +

      +
    1. An analytical solution (recall homework set 1).
    2. +
    3. The gradient can be computed analytically.
    4. +
    5. The cost function is convex which guarantees that gradient descent converges for small enough learning rates
    6. +
    + +We revisit the example from homework set 1 where we had $$ -\begin{equation*} - P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b}, -\end{equation*} +y_i = 5x_i^2 + 0.1\xi_i, \ i=1,\cdots,100 $$ -

    -with the constraint that the matrix \( \hat{A} \) is positive definite and -symmetric. This defines also the Hessian and we want it to be positive definite. +with \( x_i \in [0,1] \) chosen randomly with a uniform distribution. Additionally \( \xi_i \) represents stochastic noise chosen according to a normal distribution \( \cal {N}(0,1) \). +The linear regression model is given by +$$ +h_\beta(x) = \hat{y} = \beta_0 + \beta_1 x, +$$ + +such that +$$ +\hat{y}_i = \beta_0 + \beta_1 x_i. +$$

    @@ -310,7 +329,7 @@ symmetric. This defines also the Hessian and we want it to be positive definit

  • 28
  • 29
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs020.html b/doc/pub/Splines/html/._Splines-bs020.html index 8a8daeaa1..5636e1c24 100644 --- a/doc/pub/Splines/html/._Splines-bs020.html +++ b/doc/pub/Splines/html/._Splines-bs020.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,27 +271,29 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Steepest descent method

    +

    Gradient descent example

    -We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). -We can assume without loss of generality that +Let \( \mathbf{y} = (y_1,\cdots,y_n)^T \), \( \mathbf{\hat{y}} = (\hat{y}_1,\cdots,\hat{y}_n)^T \) and \( \beta = (\beta_0, \beta_1)^T \) + +

    +It is convenient to write \( \mathbf{\hat{y}} = X\beta \) where \( X \in \mathbb{R}^{100 \times 2} \) is the design matrix given by $$ -\begin{equation*} -\hat{x}_0=0, -\end{equation*} +X \equiv \begin{bmatrix} +1 & x_1 \\ +\vdots & \vdots \\ +1 & x_{100} & \\ +\end{bmatrix}. $$ -or consider the system +The loss function is given by $$ -\begin{equation*} -\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, -\end{equation*} +C(\beta) = ||X\beta-\mathbf{y}||^2 = ||X\beta||^2 - 2 \mathbf{y}^T X\beta + ||\mathbf{y}||^2 = \sum_{i=1}^{100} (\beta_0 + \beta_1 x_i)^2 - 2 y_i (\beta_0 + \beta_1 x_i) + y_i^2 $$ -instead. +and we want to find \( \beta \) such that \( C(\beta) \) is minimized.

    @@ -316,7 +321,7 @@ instead.

  • 29
  • 30
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs021.html b/doc/pub/Splines/html/._Splines-bs021.html index fcb22ec5f..a18f1a3d8 100644 --- a/doc/pub/Splines/html/._Splines-bs021.html +++ b/doc/pub/Splines/html/._Splines-bs021.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,33 +273,17 @@ MathJax.Hub.Config({ -

    Steepest descent method

    -
    -
    -

    -One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form -$$ -\begin{equation*} - f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. -\end{equation*} -$$ - -This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition) -to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), -which equals -$$ -\begin{equation*} -\hat{A}\hat{x}_0-\hat{b}, -\end{equation*} -$$ - -and -\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). +

    The derivative of the cost/loss function

    -

    -
    +Computing \( \partial C(\beta) / \partial \beta_0 \) and \( \partial C(\beta) / \partial \beta_1 \) we can show that the gradient can be written as +$$ +\nabla_{\beta} C(\beta) = (\partial C(\beta) / \partial \beta_0, \partial C(\beta) / \partial \beta_1)^T = 2\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\ +\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\ +\end{bmatrix} = 2X^T(X\beta - \mathbf{y}), +$$ +where \( X \) is the design matrix defined above.

    @@ -324,7 +311,7 @@ and

  • 30
  • 31
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs022.html b/doc/pub/Splines/html/._Splines-bs022.html index 70184e28a..90a6e7d4c 100644 --- a/doc/pub/Splines/html/._Splines-bs022.html +++ b/doc/pub/Splines/html/._Splines-bs022.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,46 +273,16 @@ MathJax.Hub.Config({ -

    Final expressions

    -
    -
    -

    -We can compute the residual iteratively as +

    The Hessian matrix

    +The Hessian matrix of \( C(\beta) \) is given by $$ -\begin{equation*} -\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, - \end{equation*} +\hat{H} \equiv \begin{bmatrix} +\frac{\partial^2 C(\beta)}{\partial \beta_0^2} & \frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} \\ +\frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} & \frac{\partial^2 C(\beta)}{\partial \beta_1^2} & \\ +\end{bmatrix} = 2X^T X. $$ -which equals -$$ -\begin{equation*} -\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k), - \end{equation*} -$$ - -or -$$ -\begin{equation*} -(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k, - \end{equation*} -$$ - -which gives - -$$ -\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k} -$$ - -leading to the iterative scheme -$$ -\begin{equation*} -\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k}, - \end{equation*} -$$ -
    -
    - +This result implies that \( C(\beta) \) is a convex function since the matrix \( X^T X \) always is positive semi-definite.

    @@ -337,7 +310,7 @@ $$

  • 31
  • 32
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs023.html b/doc/pub/Splines/html/._Splines-bs023.html index 5ab3a1f81..339e17063 100644 --- a/doc/pub/Splines/html/._Splines-bs023.html +++ b/doc/pub/Splines/html/._Splines-bs023.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,8 +273,44 @@ MathJax.Hub.Config({ -

    Code examples for steepest descent

    +

    Simple program

    +

    +We can now write a program that minimizes \( C(\beta) \) using the gradient descent method with a constant learning rate \( \gamma \) according to +$$ +\beta_{k+1} = \beta_k - \gamma \nabla_\beta C(\beta_k), \ k=0,1,\cdots +$$ + +

    +We can use the expression we computed for the gradient and let use a +\( \beta_0 \) be chosen randomly and let \( \gamma = 0.001 \). Stop iterating +when \( ||\nabla_\beta C(\beta_k) || \leq \epsilon = 10^{-8} \). + +

    +And finally we can compare our solution for \( \beta \) with the analytic result given by +\( \beta= (X^TX)^{-1} X^T \mathbf{y} \). +

    + + +

    import numpy as np
    +
    +"""
    +The following setup is just a suggestion, feel free to write it the way you like.
    +"""
    +
    +#Setup problem described in the exercise
    +N  = 100 #Nr of datapoints
    +M  = 2 #Nr of features
    +x  = np.random.rand(N) #Uniformly generated x-values in [0,1]
    +y  = 5*x**2 + 0.1*np.random.randn(N)
    +X  = np.c_[np.ones(N),x] #Construct design matrix
    +
    +#Compute beta according to normal equations to compare with GD solution
    +Xt_X_inv = np.linalg.inv(np.dot(X.T,X))
    +Xt_y     = np.dot(X.transpose(),y)
    +beta_NE = np.dot(Xt_X_inv,Xt_y)
    +print(beta_NE)
    +

    @@ -298,7 +337,7 @@ MathJax.Hub.Config({

  • 32
  • 33
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs024.html b/doc/pub/Splines/html/._Splines-bs024.html index 522810e59..347e15e3a 100644 --- a/doc/pub/Splines/html/._Splines-bs024.html +++ b/doc/pub/Splines/html/._Splines-bs024.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,44 +273,52 @@ MathJax.Hub.Config({ -

    Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come

    -
    -
    -

    +

    Gradient Descent Example

    + +

    +Another simple example is here

    - -

    #include <cmath>
    -#include <iostream>
    -#include <fstream>
    -#include <iomanip>
    -#include "vectormatrixclass.h"
    -using namespace  std;
    -//   Main function begins here
    -int main(int  argc, char * argv[]){
    -  int dim = 2;
    -  Vector x(dim),xsd(dim), b(dim),x0(dim);
    -  Matrix A(dim,dim);
    +
    +
    # Importing various packages
    +from random import random, seed
    +import numpy as np
    +import matplotlib.pyplot as plt
    +from mpl_toolkits.mplot3d import Axes3D
    +from matplotlib import cm
    +from matplotlib.ticker import LinearLocator, FormatStrFormatter
    +import sys
     
    -  // Set our initial guess
    -  x0(0) = x0(1) = 0;
    -  // Set the matrix
    -  A(0,0) =  3;    A(1,0) =  2;   A(0,1) =  2;   A(1,1) =  6;
    -  b(0) = 2; b(1) = -8;
    -  cout << "The Matrix A that we are using: " << endl;
    -  A.Print();
    -  cout << endl;
    -  xsd = SteepestDescent(A,b,x0);
    -  cout << "The approximate solution using Steepest Descent is: " << endl;
    -  xsd.Print();
    -  cout << endl;
    -}
    +x = 2*np.random.rand(100,1)
    +y = 4+3*x+np.random.randn(100,1)
    +
    +xb = np.c_[np.ones((100,1)), x]
    +beta_linreg = np.linalg.inv(xb.T.dot(xb)).dot(xb.T).dot(y)
    +print(beta_linreg)
    +beta = np.random.randn(2,1)
    +
    +eta = 0.1
    +Niterations = 1000
    +m = 100
    +
    +for iter in range(Niterations):
    +    gradients = 2.0/m*xb.T.dot(xb.dot(beta)-y)
    +    beta -= eta*gradients
    +
    +print(beta)
    +xnew = np.array([[0],[2]])
    +xbnew = np.c_[np.ones((2,1)), xnew]
    +ypredict = xbnew.dot(beta)
    +ypredict2 = xbnew.dot(beta_linreg)
    +plt.plot(xnew, ypredict, "r-")
    +plt.plot(xnew, ypredict2, "b-")
    +plt.plot(x, y ,'ro')
    +plt.axis([0,2.0,0, 15.0])
    +plt.xlabel(r'$x$')
    +plt.ylabel(r'$y$')
    +plt.title(r'Gradient descent example')
    +plt.show()
     
    -

    -

    -
    - -

    @@ -334,7 +345,7 @@ MathJax.Hub.Config({

  • 33
  • 34
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs025.html b/doc/pub/Splines/html/._Splines-bs025.html index 9af6a7906..3ef39008e 100644 --- a/doc/pub/Splines/html/._Splines-bs025.html +++ b/doc/pub/Splines/html/._Splines-bs025.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,40 +273,27 @@ MathJax.Hub.Config({ -

    The routine for the steepest descent method

    -
    -
    -

    +

    And a corresponding example using scikit-learn

    +

    - -

    Vector SteepestDescent(Matrix A, Vector b, Vector x0){
    -  int IterMax, i;
    -  int dim = x0.Dimension();
    -  const double tolerance = 1.0e-14;
    -  Vector x(dim),f(dim),z(dim);
    -  double c,alpha,d;
    -  IterMax = 30;
    -  x = x0;
    -  r = A*x-b;
    -  i = 0;
    -  while (i <= IterMax){
    -    z = A*r;
    -    c = dot(r,r);
    -    alpha = c/dot(r,z);
    -    x = x - alpha*r;
    -    r =  A*x-b;
    -    if(sqrt(dot(r,r)) < tolerance) break;
    -    i++;
    -  }
    -  return x;
    -}
    +
    +
    # Importing various packages
    +from random import random, seed
    +import numpy as np
    +import matplotlib.pyplot as plt
    +from sklearn.linear_model import SGDRegressor
    +
    +x = 2*np.random.rand(100,1)
    +y = 4+3*x+np.random.randn(100,1)
    +
    +xb = np.c_[np.ones((100,1)), x]
    +beta_linreg = np.linalg.inv(xb.T.dot(xb)).dot(xb.T).dot(y)
    +print(beta_linreg)
    +sgdreg = SGDRegressor(n_iter = 50, penalty=None, eta0=0.1)
    +sgdreg.fit(x,y.ravel())
    +print(sgdreg.intercept_, sgdreg.coef_)
     
    -

    -

    -
    - -

    @@ -330,7 +320,7 @@ MathJax.Hub.Config({

  • 34
  • 35
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs026.html b/doc/pub/Splines/html/._Splines-bs026.html index fa478fd15..167070dc1 100644 --- a/doc/pub/Splines/html/._Splines-bs026.html +++ b/doc/pub/Splines/html/._Splines-bs026.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,73 +271,61 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Steepest descent example

    +

    Gradient descent and Ridge

    + +

    +We have also discussed Ridge regression where the loss function contains a regularized given by the \( L_2 \) norm of \( \beta \), +$$ +C_{\text{ridge}}(\beta) = ||X\beta -\mathbf{y}||^2 + \lambda ||\beta||^2, \ \lambda \geq 0. +$$ + +

    +In order to minimize \( C_{\text{ridge}}(\beta) \) using GD we only have adjust the gradient as follows +$$ +\nabla_\beta C_{\text{ridge}}(\beta) = 2\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\ +\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\ +\end{bmatrix} + 2\lambda\begin{bmatrix} \beta_0 \\ \beta_1\end{bmatrix} = 2 (X^T(X\beta - \mathbf{y})+\lambda \beta). +$$ + +

    +We can now extend our program to minimize \( C_{\text{ridge}}(\beta) \) using gradient descent and compare with the analytical solution given by +$$ +\beta_{\text{ridge}} = \left(X^T X + \lambda I_{2 \times 2} \right)^{-1} X^T \mathbf{y}, +$$ + +for \( \lambda = {0,1,10,50,100} \) (\( \lambda = 0 \) corresponds to ordinary least squares). +We can then compute \( ||\beta_{\text{ridge}}|| \) for each \( \lambda \).

    import numpy as np
    -import numpy.linalg as la
     
    -import scipy.optimize as sopt
    +"""
    +The following setup is just a suggestion, feel free to write it the way you like.
    +"""
     
    -import matplotlib.pyplot as pt
    -from mpl_toolkits.mplot3d import axes3d
    +#Setup problem described in the exercise
    +N  = 100 #Nr of datapoints
    +M  = 2   #Nr of features
    +x  = np.random.rand(N)
    +y  = 5*x**2 + 0.1*np.random.randn(N)
     
    -def f(x):
    -    return 0.5*x[0]**2 + 2.5*x[1]**2
     
    -def df(x):
    -    return np.array([x[0], 5*x[1]])
    +#Compute analytic beta for Ridge regression 
    +X    = np.c_[np.ones(N),x]
    +XT_X = np.dot(X.T,X)
     
    -fig = pt.figure()
    -ax = fig.gca(projection="3d")
    +l  = 0.1 #Ridge parameter lambda
    +Id = np.eye(XT_X.shape[0])
     
    -xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]
    -fmesh = f(np.array([xmesh, ymesh]))
    -ax.plot_surface(xmesh, ymesh, fmesh)
    -
    -

    -And then as countor plot -

    +Z = np.linalg.inv(XT_X+l*Id) +beta_ridge = np.dot(Z,np.dot(X.T,y)) - -

    pt.axis("equal")
    -pt.contour(xmesh, ymesh, fmesh)
    -guesses = [np.array([2, 2./5])]
    -
    -

    -Find guesses -

    - - -

    x = guesses[-1]
    -s = -df(x)
    -
    -

    -Run it! -

    - - -

    def f1d(alpha):
    -    return f(x + alpha*s)
    -
    -alpha_opt = sopt.golden(f1d)
    -next_guess = x + alpha_opt * s
    -guesses.append(next_guess)
    -print(next_guess)
    -
    -

    -What happened? -

    - - -

    pt.axis("equal")
    -pt.contour(xmesh, ymesh, fmesh, 50)
    -it_array = np.array(guesses)
    -pt.plot(it_array.T[0], it_array.T[1], "x-")
    +print(beta_ridge)
    +print(np.linalg.norm(beta_ridge)) #||beta||
     

    @@ -362,7 +353,7 @@ pt.plot(it_array35

  • 36
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs027.html b/doc/pub/Splines/html/._Splines-bs027.html index 6cfa4e6a0..ab70f5a3d 100644 --- a/doc/pub/Splines/html/._Splines-bs027.html +++ b/doc/pub/Splines/html/._Splines-bs027.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,33 +273,20 @@ MathJax.Hub.Config({ -

    Conjugate gradient method

    -
    -
    -

    -In the CG method we define so-called conjugate directions and two vectors -\( \hat{s} \) and \( \hat{t} \) -are said to be -conjugate if -$$ -\begin{equation*} -\hat{s}^T\hat{A}\hat{t}= 0. -\end{equation*} -$$ +

    Stochastic Gradient Descent

    -The philosophy of the CG method is to perform searches in various conjugate directions -of our vectors \( \hat{x}_i \) obeying the above criterion, namely -$$ -\begin{equation*} -\hat{x}_i^T\hat{A}\hat{x}_j= 0. -\end{equation*} -$$ - -Two vectors are conjugate if they are orthogonal with respect to -this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \). -
    -
    +

    +Stochastic gradient descent (SGD) and variants thereof address some of +the shortcomings of the Gradient descent method discussed above. +

    +The underlying idea of SGD comes from the observation that the cost +function, which we want to minimize, can almost always be written as a +sum over \( n \) data points \( \{\mathbf{x}_i\}_{i=1}^n \), +$$ +C(\mathbf{\beta}) = \sum_{i=1}^n c_i(\mathbf{x}_i, +\mathbf{\beta}). +$$

    @@ -324,7 +314,7 @@ this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is

  • 36
  • 37
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs028.html b/doc/pub/Splines/html/._Splines-bs028.html index 13e6a5758..12026c49b 100644 --- a/doc/pub/Splines/html/._Splines-bs028.html +++ b/doc/pub/Splines/html/._Splines-bs028.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,21 +273,22 @@ MathJax.Hub.Config({ -

    Conjugate gradient method

    -
    -
    -

    -An example is given by the eigenvectors of the matrix +

    Computation of gradients

    + +

    +This in turn means that the gradient can be +computed as a sum over \( i \)-gradients $$ -\begin{equation*} -\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j, -\end{equation*} +\nabla_\beta C(\mathbf{\beta}) = \sum_i^n \nabla_\beta c_i(\mathbf{x}_i, +\mathbf{\beta}). $$ -which is zero unless \( i=j \). -

    -
    - +

    +Stochasticity/randomness is introduced by only taking the +gradient on a subset of the data called minibatches. If there are \( n \) +data points and the size of each minibatch is \( M \), there will be \( n/M \) +minibatches. We denote these minibatches by \( B_k \) where +\( k=1,\cdots,n/M \).

    @@ -312,7 +316,7 @@ which is zero unless \( i=j \).

  • 37
  • 38
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs029.html b/doc/pub/Splines/html/._Splines-bs029.html index ca54432bb..f64b677ff 100644 --- a/doc/pub/Splines/html/._Splines-bs029.html +++ b/doc/pub/Splines/html/._Splines-bs029.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,30 +273,26 @@ MathJax.Hub.Config({ -

    Conjugate gradient method

    -
    -
    -

    -Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size -\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector -$$ -\begin{equation*} -\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}. -\end{equation*} -$$ - -We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions. -Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution -$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely +

    SGD example

    +As an example, suppose we have \( 10 \) data points \( (\mathbf{x}_1,\cdots, \mathbf{x}_{10}) \) +and we choose to have \( M=5 \) minibathces, +then each minibatch contains two data points. In particular we have +\( B_1 = (\mathbf{x}_1,\mathbf{x}_2), \cdots, B_5 = +(\mathbf{x}_9,\mathbf{x}_{10}) \). Note that if you choose \( M=1 \) you +have only a single batch with all data points and on the other extreme, +you may choose \( M=n \) resulting in a minibatch for each datapoint, i.e +\( B_k = \mathbf{x}_k \). +

    +The idea is now to approximate the gradient by replacing the sum over +all data points with a sum over the data points in one the minibatches +picked at random in each gradient descent step $$ -\begin{equation*} - \hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i. -\end{equation*} +\nabla_{\beta} +C(\mathbf{\beta}) = \sum_{i=1}^n \nabla_\beta c_i(\mathbf{x}_i, +\mathbf{\beta}) \rightarrow \sum_{i \in B_k}^n \nabla_\beta +c_i(\mathbf{x}_i, \mathbf{\beta}). $$ -

    -
    -

    @@ -321,7 +320,7 @@ $$

  • 38
  • 39
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs030.html b/doc/pub/Splines/html/._Splines-bs030.html index 3a9f154ec..ff3289ddc 100644 --- a/doc/pub/Splines/html/._Splines-bs030.html +++ b/doc/pub/Splines/html/._Splines-bs030.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,35 +273,21 @@ MathJax.Hub.Config({ -

    Conjugate gradient method

    -
    -
    -

    -The coefficients are given by +

    The gradient step

    + +

    +Thus a gradient descent step now looks like $$ -\begin{equation*} - \mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. -\end{equation*} +\beta_{j+1} = \beta_j - \gamma_j \sum_{i \in B_k}^n \nabla_\beta c_i(\mathbf{x}_i, +\mathbf{\beta}) $$ -Multiplying with \( \hat{p}_k^T \) from the left gives - -$$ -\begin{equation*} - \hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b}, -\end{equation*} -$$ - -and we can define the coefficients \( \alpha_k \) as - -$$ -\begin{equation*} - \alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k} -\end{equation*} -$$ -

    -
    - +

    +where \( k \) is picked at random with equal +probability from \( [1,n/M] \). An iteration over the number of +minibathces (n/M) is commonly referred to as an epoch. Thus it is +typical to choose a number of epochs and for each epoch iterate over +the number of minibatches, as exemplified in the code below.

    @@ -326,7 +315,7 @@ $$

  • 39
  • 40
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs031.html b/doc/pub/Splines/html/._Splines-bs031.html index 1d403ba2a..a388c99de 100644 --- a/doc/pub/Splines/html/._Splines-bs031.html +++ b/doc/pub/Splines/html/._Splines-bs031.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,39 +273,34 @@ MathJax.Hub.Config({ -

    Conjugate gradient method and iterations

    -
    -
    -

    +

    Simple example code

    -If we choose the conjugate vectors \( \hat{p}_k \) carefully, -then we may not need all of them to obtain a good approximation to the solution -\( \hat{x} \). -We want to regard the conjugate gradient method as an iterative method. -This will us to solve systems where \( n \) is so large that the direct -method would take too much time. + +

    import numpy as np 
    +
    +n = 100 #100 datapoints 
    +M = 5   #size of each minibatch
    +m = int(n/M) #number of minibatches
    +n_epochs = 10 #number of epochs
    +
    +j = 0
    +for epoch in range(1,n_epochs+1):
    +    for i in range(m):
    +        k = np.random.randint(m) #Pick the k-th minibatch at random
    +        #Compute the gradient using the data in minibatch Bk
    +        #Compute new suggestion for 
    +        j += 1
    +

    -We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). -We can assume without loss of generality that -$$ -\begin{equation*} -\hat{x}_0=0, -\end{equation*} -$$ - -or consider the system -$$ -\begin{equation*} -\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, -\end{equation*} -$$ - -instead. -

    -
    - +Taking the gradient only on a subset of the data has two important +benefits. First, it introduces randomness which decreases the chance +that our opmization scheme gets stuck in a local minima. Second, if +the size of the minibatches are small relative to the number of +datapoints (\( M < n \)), the computation of the gradient is much +cheaper since we sum over the datapoints in the \( k-th \) minibatch and not +all \( n \) datapoints.

    @@ -330,7 +328,7 @@ instead.

  • 40
  • 41
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs032.html b/doc/pub/Splines/html/._Splines-bs032.html index c76b7d002..5a220170c 100644 --- a/doc/pub/Splines/html/._Splines-bs032.html +++ b/doc/pub/Splines/html/._Splines-bs032.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,33 +273,19 @@ MathJax.Hub.Config({ -

    Conjugate gradient method

    -
    -
    -

    -One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form -$$ -\begin{equation*} - f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. -\end{equation*} -$$ - -This suggests taking the first basis vector \( \hat{p}_1 \) -to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), -which equals -$$ -\begin{equation*} -\hat{A}\hat{x}_0-\hat{b}, -\end{equation*} -$$ - -and -\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). -The other vectors in the basis will be conjugate to the gradient, -hence the name conjugate gradient method. -

    -
    +

    When do we stop?

    +

    +A natural question is when do we stop the search for a new minimum? +One possibility is to compute the full gradient after a given number +of epochs and check if the norm of the gradient is smaller than some +threshold and stop if true. However, the condition that the gradient +is zero is valid also for local minima, so this would only tell us +that we are close to a local/global minimum. However, we could also +evaluate the cost function at this point, store the result and +continue the search. If the test kicks in at a later stage we can +compare the values of the cost function and keep the \( \beta \) that +gave the lowest value.

    @@ -324,7 +313,7 @@ hence the name conjugate gradient method.

  • 41
  • 42
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs033.html b/doc/pub/Splines/html/._Splines-bs033.html index 3df1a220d..a0ad7d857 100644 --- a/doc/pub/Splines/html/._Splines-bs033.html +++ b/doc/pub/Splines/html/._Splines-bs033.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,33 +273,51 @@ MathJax.Hub.Config({ -

    Conjugate gradient method

    -
    -
    -

    -Let \( \hat{r}_k \) be the residual at the \( k \)-th step: -$$ -\begin{equation*} -\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k. -\end{equation*} -$$ +

    Slightly different approach

    -Note that \( \hat{r}_k \) is the negative gradient of \( f \) at -\( \hat{x}=\hat{x}_k \), -so the gradient descent method would be to move in the direction \( \hat{r}_k \). -Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other, -so we take the direction closest to the gradient \( \hat{r}_k \) -under the conjugacy constraint. -This gives the following expression -$$ -\begin{equation*} -\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k. -\end{equation*} -$$ -
    -
    +

    +Another approach is to let the step length \( \gamma_j \) depend on the +number of epochs in such a way that it becomes very small after a +reasonable time such that we do not move at all. +

    +As an example, let \( e = 0,1,2,3,\cdots \) denote the current epoch and let \( t_0, t_1 > 0 \) be two fixed numbers. Furthermore, let \( t = e \cdot m + i \) where \( m \) is the number of minibatches and \( i=0,\cdots,m-1 \). Then the function $$\gamma_j(t; t_0, t_1) = \frac{t_0}{t+t_1} $$ goes to zero as the number of epochs gets large. I.e. we start with a step length \( \gamma_j (0; t_0, t_1) = t_0/t_1 \) which decays in time \( t \). +

    +In this way we can fix the number of epochs, compute \( \beta \) and +evaluate the cost function at the end. Repeating the computation will +give a different result since the scheme is random by design. Then we +pick the final \( \beta \) that gives the lowest value of the cost +function. + +

    + + +

    import numpy as np 
    +
    +def step_length(t,t0,t1):
    +    return t0/(t+t1)
    +
    +n = 100 #100 datapoints 
    +M = 5   #size of each minibatch
    +m = int(n/M) #number of minibatches
    +n_epochs = 500 #number of epochs
    +t0 = 1.0
    +t1 = 10
    +
    +gamma_j = t0/t1
    +j = 0
    +for epoch in range(1,n_epochs+1):
    +    for i in range(m):
    +        k = np.random.randint(m) #Pick the k-th minibatch at random
    +        #Compute the gradient using the data in minibatch Bk
    +        #Compute new suggestion for beta
    +        t = epoch*m+i
    +        gamma_j = step_length(t,t0,t1)
    +        j += 1
    +
    +print("gamma_j after %d epochs: %g" % (n_epochs,gamma_j))
    +

    @@ -323,7 +344,7 @@ $$

  • 42
  • 43
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs034.html b/doc/pub/Splines/html/._Splines-bs034.html index 7542aab15..108dbd892 100644 --- a/doc/pub/Splines/html/._Splines-bs034.html +++ b/doc/pub/Splines/html/._Splines-bs034.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,42 +273,82 @@ MathJax.Hub.Config({ -

    Conjugate gradient method

    -
    -
    -

    -We can also compute the residual iteratively as -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, - \end{equation*} -$$ +

    Program for stochastic gradient

    -which equals -$$ -\begin{equation*} -\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k), - \end{equation*} -$$ +

    -or -$$ -\begin{equation*} -(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k, - \end{equation*} -$$ + +

    # Importing various packages
    +from math import exp, sqrt
    +from random import random, seed
    +import numpy as np
    +import matplotlib.pyplot as plt
    +from sklearn.linear_model import SGDRegressor
     
    -which gives
    +x = 2*np.random.rand(100,1)
    +y = 4+3*x+np.random.randn(100,1)
     
    -$$
    -\begin{equation*}
    -\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k},
    - \end{equation*}
    -$$
    -
    -
    +xb = np.c_[np.ones((100,1)), x] +theta_linreg = np.linalg.inv(xb.T.dot(xb)).dot(xb.T).dot(y) +print("Own inversion") +print(theta_linreg) +sgdreg = SGDRegressor(n_iter = 50, penalty=None, eta0=0.1) +sgdreg.fit(x,y.ravel()) +print("sgdreg from scikit") +print(sgdreg.intercept_, sgdreg.coef_) +theta = np.random.randn(2,1) + +eta = 0.1 +Niterations = 1000 +m = 100 + +for iter in range(Niterations): + gradients = 2.0/m*xb.T.dot(xb.dot(theta)-y) + theta -= eta*gradients +print("theta frm own gd") +print(theta) + +xnew = np.array([[0],[2]]) +xbnew = np.c_[np.ones((2,1)), xnew] +ypredict = xbnew.dot(theta) +ypredict2 = xbnew.dot(theta_linreg) + + +n_epochs = 50 +t0, t1 = 5, 50 +m = 100 +def learning_schedule(t): + return t0/(t+t1) + +theta = np.random.randn(2,1) + +for epoch in range(n_epochs): + for i in range(m): + random_index = np.random.randint(m) + xi = xb[random_index:random_index+1] + yi = y[random_index:random_index+1] + gradients = 2 * xi.T.dot(xi.dot(theta)-yi) + eta = learning_schedule(epoch*m+i) + theta = theta - eta*gradients +print("theta from own sdg") +print(theta) + + + + + + +plt.plot(xnew, ypredict, "r-") +plt.plot(xnew, ypredict2, "b-") +plt.plot(x, y ,'ro') +plt.axis([0,2.0,0, 15.0]) +plt.xlabel(r'$x$') +plt.ylabel(r'$y$') +plt.title(r'Random numbers ') +plt.show() +

    @@ -332,7 +375,7 @@ $$

  • 43
  • 44
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs035.html b/doc/pub/Splines/html/._Splines-bs035.html index 490acfbac..3d7ec7df0 100644 --- a/doc/pub/Splines/html/._Splines-bs035.html +++ b/doc/pub/Splines/html/._Splines-bs035.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,44 +273,17 @@ MathJax.Hub.Config({ -

    Simple implementation of the Conjugate gradient algorithm

    -
    -
    -

    -

    +

    Using gradient descent methods, limitations

    - -
      Vector ConjugateGradient(Matrix A, Vector b, Vector x0){
    -  int dim = x0.Dimension();
    -  const double tolerance = 1.0e-14;
    -  Vector x(dim),r(dim),v(dim),z(dim);
    -  double c,t,d;
    +
      +
    • Gradient descent (GD) finds local minima of our function. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our energy function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.
    • +
    • GD is sensitive to initial conditions. One consequence of the local nature of GD is that initial conditions matter. Depending on where one starts, one will end up at a different local minima. Therefore, it is very important to think about how one initializes the training process. This is true for GD as well as more complicated variants of GD.
    • +
    • Gradients are computationally expensive to calculate for large datasets. In many cases in statistics and ML, the energy function is a sum of terms, with one term for each data point. For example, in linear regression, \( E \propto \sum_{i=1}^n (y_i - \mathbf{w}^T\cdot\mathbf{x}_i)^2 \); for logistic regression, the square error is replaced by the cross entropy. To calculate the gradient we have to sum over all \( n \) data points. Doing this at every GD step becomes extremely computationally expensive. An ingenious solution to this, is to calculate the gradients using small subsets of the data called "mini batches". This has the added benefit of introducing stochasticity into our algorithm.
    • +
    • GD is very sensitive to choices of learning rates. GD is extremely sensitive to the choice of learning rates. If the learning rate is very small, the training process take an extremely long time. For larger learning rates, GD can diverge and give poor results. Furthermore, depending on what the local landscape looks like, we have to modify the learning rates to ensure convergence. Ideally, we would adaptively choose the learning rates to match the landscape.
    • +
    • GD treats all directions in parameter space uniformly. Another major drawback of GD is that unlike Newton's method, the learning rate for GD is the same in all directions in parameter space. For this reason, the maximum learning rate is set by the behavior of the steepest direction and this can significantly slow down training. Ideally, we would like to take large steps in flat directions and small steps in steep directions. Since we are exploring rugged landscapes where curvatures change, this requires us to keep track of not only the gradient but second derivatives. The ideal scenario would be to calculate the Hessian but this proves to be too computationally expensive.
    • +
    • GD can take exponential time to escape saddle points, even with random initialization. As we mentioned, GD is extremely sensitive to initial condition since it determines the particular local minimum GD would eventually reach. However, even with a good initialization scheme, through the introduction of randomness, GD can still take exponential time to escape saddle points.
    • +
    - x = x0; - r = b - A*x; - v = r; - c = dot(r,r); - int i = 0; IterMax = dim; - while(i <= IterMax){ - z = A*v; - t = c/dot(v,z); - x = x + t*v; - r = r - t*z; - d = dot(r,r); - if(sqrt(d) < tolerance) - break; - v = r + (d/c)*v; - c = d; i++; - } - return x; -} -
    -

    -

    -
    - - -

      @@ -333,7 +309,7 @@ MathJax.Hub.Config({
    • 44
    • 45
    • ...
    • -
    • 72
    • +
    • 73
    • »
    diff --git a/doc/pub/Splines/html/._Splines-bs036.html b/doc/pub/Splines/html/._Splines-bs036.html index 0693decbf..10e862a83 100644 --- a/doc/pub/Splines/html/._Splines-bs036.html +++ b/doc/pub/Splines/html/._Splines-bs036.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,38 +273,25 @@ MathJax.Hub.Config({ -

    Broyden–Fletcher–Goldfarb–Shanno algorithm

    -
    -
    -

    -The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take. +

    Momentum based GD

    -The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage. - -

    -The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation +The stochastic gradient descent (SGD) is almost always used with a momentum or inertia term that serves as a memory of the direction we are moving in parameter space. This is typically +implemented as follows $$ -B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}), +\begin{align} +\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t) \nonumber \\ +\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t}, +\tag{2} +\end{align} $$ -

    -where \( B_{k} \) is an approximation to the Hessian matrix, which is -updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \) -is the gradient of the function -evaluated at \( x_k \). -A line search in the direction \( p_k \) is then used to -find the next point \( x_{k+1} \) by minimising +where we have introduced a momentum parameter \( \gamma \), with \( 0\le\gamma\le 1 \), and for brevity we dropped the explicit notation to indicate the gradient is to be taken over a different mini-batch at each step. We call this algorithm gradient descent with momentum (GDM). From these equations, it is clear that \( \mathbf{v}_t \) is a running average of recently encountered gradients and \( (1-\gamma)^{-1} \) sets the characteristic time scale for the memory used in the averaging procedure. Consistent with this, when \( \gamma=0 \), this just reduces down to ordinary SGD as discussed earlier. An equivalent way of writing the updates is $$ -f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}), +\Delta \boldsymbol{\theta}_{t+1} = \gamma \Delta \boldsymbol{\theta}_t -\ \eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t), $$ -over the scalar \( \alpha > 0 \). - -

    -

    -
    - +where we have defined \( \Delta \boldsymbol{\theta}_{t}= \boldsymbol{\theta}_t-\boldsymbol{\theta}_{t-1} \).

    @@ -329,7 +319,7 @@ over the scalar \( \alpha > 0 \).

  • 45
  • 46
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs037.html b/doc/pub/Splines/html/._Splines-bs037.html index 7fa16c76b..ff2237aef 100644 --- a/doc/pub/Splines/html/._Splines-bs037.html +++ b/doc/pub/Splines/html/._Splines-bs037.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,36 +271,25 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Revisiting our first homework

    +

    More on momentum based approaches

    -We will use linear regression as a case study for the gradient descent -methods. Linear regression is a great test case for the gradient -descent methods discussed in the lectures since it has several -desirable properties such as: - -

      -
    1. An analytical solution (recall homework set 1).
    2. -
    3. The gradient can be computed analytically.
    4. -
    5. The cost function is convex which guarantees that gradient descent converges for small enough learning rates
    6. -
    - -We revisit the example from homework set 1 where we had +Let us try to get more intuition from these equations. It is helpful to consider a simple physical analogy with a particle of mass \( m \) moving in a viscous medium with drag coefficient \( \mu \) and potential +\( E(\mathbf{w}) \). If we denote the particle's position by \( \mathbf{w} \), then its motion is described by $$ -y_i = 5x_i^2 + 0.1\xi_i, \ i=1,\cdots,100 +m {d^2 \mathbf{w} \over dt^2} + \mu {d \mathbf{w} \over dt }= -\nabla_w E(\mathbf{w}). $$ -with \( x_i \in [0,1] \) chosen randomly with a uniform distribution. Additionally \( \xi_i \) represents stochastic noise chosen according to a normal distribution \( \cal {N}(0,1) \). -The linear regression model is given by +We can discretize this equation in the usual way to get $$ -h_\beta(x) = \hat{y} = \beta_0 + \beta_1 x, +m { \mathbf{w}_{t+\Delta t}-2 \mathbf{w}_{t} +\mathbf{w}_{t-\Delta t} \over (\Delta t)^2}+\mu {\mathbf{w}_{t+\Delta t}- \mathbf{w}_{t} \over \Delta t} = -\nabla_w E(\mathbf{w}). $$ -such that +Rearranging this equation, we can rewrite this as $$ -\hat{y}_i = \beta_0 + \beta_1 x_i. +\Delta \mathbf{w}_{t +\Delta t}= - { (\Delta t)^2 \over m +\mu \Delta t} \nabla_w E(\mathbf{w})+ {m \over m +\mu \Delta t} \Delta \mathbf{w}_t. $$

    @@ -326,7 +318,7 @@ $$

  • 46
  • 47
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs038.html b/doc/pub/Splines/html/._Splines-bs038.html index 91712272d..9001d9643 100644 --- a/doc/pub/Splines/html/._Splines-bs038.html +++ b/doc/pub/Splines/html/._Splines-bs038.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,29 +271,34 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Gradient descent example

    +

    Momentum parameter

    +Notice that this equation is identical to previous one if we identify the position of the particle, \( \mathbf{w} \), with the parameters \( \boldsymbol{\theta} \). This allows +us to identify the momentum parameter and learning rate with the mass of the particle and the viscous drag as: +$$ +\gamma= {m \over m +\mu \Delta t }, \qquad \eta = {(\Delta t)^2 \over m +\mu \Delta t}. +$$ + +Thus, as the name suggests, the momentum parameter is proportional to the mass of the particle and effectively provides inertia. Furthermore, in the large viscosity/small learning rate limit, our memory time scales as \( (1-\gamma)^{-1} \approx m/(\mu \Delta t) \).

    -Let \( \mathbf{y} = (y_1,\cdots,y_n)^T \), \( \mathbf{\hat{y}} = (\hat{y}_1,\cdots,\hat{y}_n)^T \) and \( \beta = (\beta_0, \beta_1)^T \) +Why is momentum useful? SGD momentum helps the gradient descent algorithm gain speed in directions with persistent but small gradients even in the presence of stochasticity, while suppressing oscillations in high-curvature directions. This becomes especially important in situations where the landscape is shallow and flat in some directions and narrow and steep in others. It has been argued that first-order methods (with appropriate initial conditions) can perform comparable to more expensive second order methods, especially in the context of complex deep learning models.

    -It is convenient to write \( \mathbf{\hat{y}} = X\beta \) where \( X \in \mathbb{R}^{100 \times 2} \) is the design matrix given by +These beneficial properties of momentum can sometimes become even more pronounced by using a slight modification of the classical momentum algorithm called Nesterov Accelerated Gradient (NAG). + +

    +In the NAG algorithm, rather than calculating the gradient at the current parameters, \( \nabla_\theta E(\boldsymbol{\theta}_t) \), one calculates the gradient at the expected value of the parameters given our current momentum, \( \nabla_\theta E(\boldsymbol{\theta}_t +\gamma \mathbf{v}_{t-1}) \). This yields the NAG update rule $$ -X \equiv \begin{bmatrix} -1 & x_1 \\ -\vdots & \vdots \\ -1 & x_{100} & \\ -\end{bmatrix}. +\begin{align} +\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t +\gamma \mathbf{v}_{t-1}) \nonumber \\ +\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t}. +\tag{3} +\end{align} $$ -The loss function is given by -$$ -C(\beta) = ||X\beta-\mathbf{y}||^2 = ||X\beta||^2 - 2 \mathbf{y}^T X\beta + ||\mathbf{y}||^2 = \sum_{i=1}^{100} (\beta_0 + \beta_1 x_i)^2 - 2 y_i (\beta_0 + \beta_1 x_i) + y_i^2 -$$ - -and we want to find \( \beta \) such that \( C(\beta) \) is minimized. +One of the major advantages of NAG is that it allows for the use of a larger learning rate than GDM for the same choice of \( \gamma \).

    @@ -318,7 +326,7 @@ and we want to find \( \beta \) such that \( C(\beta) \) is minimized.

  • 47
  • 48
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs039.html b/doc/pub/Splines/html/._Splines-bs039.html index 2b0dc7059..cd4450ecd 100644 --- a/doc/pub/Splines/html/._Splines-bs039.html +++ b/doc/pub/Splines/html/._Splines-bs039.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,17 +273,27 @@ MathJax.Hub.Config({ -

    The derivative of the cost/loss function

    +

    Second moment of the gradient

    -Computing \( \partial C(\beta) / \partial \beta_0 \) and \( \partial C(\beta) / \partial \beta_1 \) we can show that the gradient can be written as -$$ -\nabla_{\beta} C(\beta) = (\partial C(\beta) / \partial \beta_0, \partial C(\beta) / \partial \beta_1)^T = 2\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\ -\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\ -\end{bmatrix} = 2X^T(X\beta - \mathbf{y}), -$$ +In stochastic gradient descent, with and without momentum, we still +have to specify a schedule for tuning the learning rates \( \eta_t \) +as a function of time. As discussed in the context of Newton's +method, this presents a number of dilemmas. The learning rate is +limited by the steepest direction which can change depending on the +current position in the landscape. To circumvent this problem, ideally +our algorithm would keep track of curvature and take large steps in +shallow, flat directions and small steps in steep, narrow directions. +Second-order methods accomplish this by calculating or approximating +the Hessian and normalizing the learning rate by the +curvature. However, this is very computationally expensive for +extremely large models. Ideally, we would like to be able to +adaptively change the step size to match the landscape without paying +the steep computational price of calculating or approximating +Hessians. -where \( X \) is the design matrix defined above. +

    +Recently, a number of methods have been introduced that accomplish this by tracking not only the gradient, but also the second moment of the gradient. These methods include AdaGrad, AdaDelta, RMS-Prop, and ADAM.

    @@ -308,7 +321,7 @@ where \( X \) is the design matrix defined above.

  • 48
  • 49
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs040.html b/doc/pub/Splines/html/._Splines-bs040.html index c9339c086..5f7eab6eb 100644 --- a/doc/pub/Splines/html/._Splines-bs040.html +++ b/doc/pub/Splines/html/._Splines-bs040.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,16 +273,20 @@ MathJax.Hub.Config({ -

    The Hessian matrix

    -The Hessian matrix of \( C(\beta) \) is given by +

    RMS prop

    + +

    +In RMS prop, in addition to keeping a running average of the first moment of the gradient, we also keep track of the second moment denoted by \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}_t^2] \). The update rule for RMS prop is given by $$ -\hat{H} \equiv \begin{bmatrix} -\frac{\partial^2 C(\beta)}{\partial \beta_0^2} & \frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} \\ -\frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} & \frac{\partial^2 C(\beta)}{\partial \beta_1^2} & \\ -\end{bmatrix} = 2X^T X. +\begin{align} +\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta}) +\tag{4}\\ +\mathbf{s}_t &=\beta \mathbf{s}_{t-1} +(1-\beta)\mathbf{g}_t^2 \nonumber \\ +\boldsymbol{\theta}_{t+1}&=&\boldsymbol{\theta}_t - \eta_t { \mathbf{g}_t \over \sqrt{\mathbf{s}_t +\epsilon}}, \nonumber +\end{align} $$ -This result implies that \( C(\beta) \) is a convex function since the matrix \( X^T X \) always is positive semi-definite. +where \( \beta \) controls the averaging time of the second moment and is typically taken to be about \( \beta=0.9 \), \( \eta_t \) is a learning rate typically chosen to be \( 10^{-3} \), and \( \epsilon\sim 10^{-8} \) is a small regularization constant to prevent divergences. Multiplication and division by vectors is understood as an element-wise operation. It is clear from this formula that the learning rate is reduced in directions where the norm of the gradient is consistently large. This greatly speeds up the convergence by allowing us to use a larger learning rate for flat directions.

    @@ -307,7 +314,7 @@ This result implies that \( C(\beta) \) is a convex function since the matrix \(

  • 49
  • 50
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs041.html b/doc/pub/Splines/html/._Splines-bs041.html index 854e2a263..fb10e30b8 100644 --- a/doc/pub/Splines/html/._Splines-bs041.html +++ b/doc/pub/Splines/html/._Splines-bs041.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,44 +273,31 @@ MathJax.Hub.Config({ -

    Simple program

    +

    ADAM optimizer

    -We can now write a program that minimizes \( C(\beta) \) using the gradient descent method with a constant learning rate \( \gamma \) according to +A related algorithm is the ADAM optimizer. In ADAM, we keep a running average of both the first and second moment of the gradient and use this information to adaptively change the learning rate for different parameters. In addition to keeping a running average of the first and second moments of the gradient (i.e. \( \mathbf{m}_t=\mathbb{E}[\mathbf{g}_t] \) and \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}^2_t] \), respectively), ADAM performs an additional bias correction to account for the fact that we are estimating the first two moments of the gradient using a running average (denoted by the hats in the update rule below). The update rule for ADAM is given by (where multiplication and division are once again understood to be element-wise operations below) $$ -\beta_{k+1} = \beta_k - \gamma \nabla_\beta C(\beta_k), \ k=0,1,\cdots +\begin{align} +\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta}) +\tag{5}\\ +\mathbf{m}_t &= \beta_1 \mathbf{m}_{t-1} + (1-\beta_1) \mathbf{g}_t \nonumber \\ +\mathbf{s}_t &=\beta_2 \mathbf{s}_{t-1} +(1-\beta_2)\mathbf{g}_t^2 \nonumber \\ +\hat{\mathbf{m}}_t&={\mathbf{m}_t \over 1-\beta_1^t} \nonumber \\ +\hat{\mathbf{s}}_t &={\mathbf{s}_t \over1-\beta_2^t} \nonumber \\ +\boldsymbol{\theta}_{t+1}&=\boldsymbol{\theta}_t - \eta_t { \hat{\mathbf{m}}_t \over \sqrt{\hat{\mathbf{s}}_t} +\epsilon}, \nonumber \\ +\tag{6} +\end{align} $$ -

    -We can use the expression we computed for the gradient and let use a -\( \beta_0 \) be chosen randomly and let \( \gamma = 0.001 \). Stop iterating -when \( ||\nabla_\beta C(\beta_k) || \leq \epsilon = 10^{-8} \). +where \( \beta_1 \) and \( \beta_2 \) set the memory lifetime of the first and second moment and are typically taken to be \( 0.9 \) and \( 0.99 \) respectively, and \( \eta \) and \( \epsilon \) are identical to RMSprop.

    -And finally we can compare our solution for \( \beta \) with the analytic result given by -\( \beta= (X^TX)^{-1} X^T \mathbf{y} \). -

    +Like in RMSprop, the effective step size of a parameter depends on the magnitude of its gradient squared. To understand this better, let us rewrite this expression in terms of the variance \( \boldsymbol{\sigma}_t^2 = \hat{\mathbf{s}}_t - (\hat{\mathbf{m}}_t)^2 \). Consider a single parameter \( \theta_t \). The update rule for this parameter is given by +$$ +\Delta \theta_{t+1}= -\eta_t { \hat{m}_t \over \sqrt{\sigma_t^2 + m_t^2 }+\epsilon}. +$$ - -

    import numpy as np
    -
    -"""
    -The following setup is just a suggestion, feel free to write it the way you like.
    -"""
    -
    -#Setup problem described in the exercise
    -N  = 100 #Nr of datapoints
    -M  = 2 #Nr of features
    -x  = np.random.rand(N) #Uniformly generated x-values in [0,1]
    -y  = 5*x**2 + 0.1*np.random.randn(N)
    -X  = np.c_[np.ones(N),x] #Construct design matrix
    -
    -#Compute beta according to normal equations to compare with GD solution
    -Xt_X_inv = np.linalg.inv(np.dot(X.T,X))
    -Xt_y     = np.dot(X.transpose(),y)
    -beta_NE = np.dot(Xt_X_inv,Xt_y)
    -print(beta_NE)
    -

    @@ -334,7 +324,7 @@ beta_NE = np.50

  • 51
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs042.html b/doc/pub/Splines/html/._Splines-bs042.html index 1af463424..2d89207eb 100644 --- a/doc/pub/Splines/html/._Splines-bs042.html +++ b/doc/pub/Splines/html/._Splines-bs042.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,52 +273,17 @@ MathJax.Hub.Config({ -

    Gradient Descent Example

    +

    Practical tips

    -

    -Another simple example is here -

    +

      +
    • Randomize the data when making mini-batches. It is always important to randomly shuffle the data when forming mini-batches. Otherwise, the gradient descent method can fit spurious correlations resulting from the order in which data is presented.
    • +
    • Transform your inputs. Learning becomes difficult when our landscape has a mixture of steep and flat directions. One simple trick for minimizing these situations is to standardize the data by subtracting the mean and normalizing the variance of input variables. Whenever possible, also decorrelate the inputs. To understand why this is helpful, consider the case of linear regression. It is easy to show that for the squared error cost function, the Hessian of the energy matrix is just the correlation matrix between the inputs. Thus, by standardizing the inputs, we are ensuring that the landscape looks homogeneous in all directions in parameter space. Since most deep networks can be viewed as linear transformations followed by a non-linearity at each layer, we expect this intuition to hold beyond the linear case.
    • +
    • Monitor the out-of-sample performance. Always monitor the performance of your model on a validation set (a small portion of the training data that is held out of the training process to serve as a proxy for the test set. If the validation error starts increasing, then the model is beginning to overfit. Terminate the learning process. This early stopping significantly improves performance in many settings.
    • +
    • Adaptive optimization methods don't always have good generalization. Recent studies have shown that adaptive methods such as ADAM, RMSPorp, and AdaGrad tend to have poor generalization compared to SGD or SGD with momentum, particularly in the high-dimensional limit (i.e. the number of parameters exceeds the number of data points). Although it is not clear at this stage why these methods perform so well in training deep neural networks, simpler procedures like properly-tuned SGD may work as well or better in these applications.
    • +
    - -
    # Importing various packages
    -from random import random, seed
    -import numpy as np
    -import matplotlib.pyplot as plt
    -from mpl_toolkits.mplot3d import Axes3D
    -from matplotlib import cm
    -from matplotlib.ticker import LinearLocator, FormatStrFormatter
    -import sys
    +Geron's text, see chapter 11, has several interesting discussions.
     
    -x = 2*np.random.rand(100,1)
    -y = 4+3*x+np.random.randn(100,1)
    -
    -xb = np.c_[np.ones((100,1)), x]
    -beta_linreg = np.linalg.inv(xb.T.dot(xb)).dot(xb.T).dot(y)
    -print(beta_linreg)
    -beta = np.random.randn(2,1)
    -
    -eta = 0.1
    -Niterations = 1000
    -m = 100
    -
    -for iter in range(Niterations):
    -    gradients = 2.0/m*xb.T.dot(xb.dot(beta)-y)
    -    beta -= eta*gradients
    -
    -print(beta)
    -xnew = np.array([[0],[2]])
    -xbnew = np.c_[np.ones((2,1)), xnew]
    -ypredict = xbnew.dot(beta)
    -ypredict2 = xbnew.dot(beta_linreg)
    -plt.plot(xnew, ypredict, "r-")
    -plt.plot(xnew, ypredict2, "b-")
    -plt.plot(x, y ,'ro')
    -plt.axis([0,2.0,0, 15.0])
    -plt.xlabel(r'$x$')
    -plt.ylabel(r'$y$')
    -plt.title(r'Gradient descent example')
    -plt.show()
    -

    @@ -342,7 +310,7 @@ plt.show()

  • 51
  • 52
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs043.html b/doc/pub/Splines/html/._Splines-bs043.html index 85270e43f..491e9bc4a 100644 --- a/doc/pub/Splines/html/._Splines-bs043.html +++ b/doc/pub/Splines/html/._Splines-bs043.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,26 +273,57 @@ MathJax.Hub.Config({ -

    And a corresponding example using scikit-learn

    +

    Automatic differentiation

    +Python has tools for so-called automatic differentiation. +Consider the following example +$$ +f(x) = \sin\left(2\pi x + x^2\right) +$$ + +which has the following derivative +$$ +f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) +$$ + +Using autograd we have

    -

    # Importing various packages
    -from random import random, seed
    -import numpy as np
    -import matplotlib.pyplot as plt
    -from sklearn.linear_model import SGDRegressor
    +
    import autograd.numpy as np
     
    -x = 2*np.random.rand(100,1)
    -y = 4+3*x+np.random.randn(100,1)
    +# To do elementwise differentiation:
    +from autograd import elementwise_grad as egrad 
     
    -xb = np.c_[np.ones((100,1)), x]
    -beta_linreg = np.linalg.inv(xb.T.dot(xb)).dot(xb.T).dot(y)
    -print(beta_linreg)
    -sgdreg = SGDRegressor(n_iter = 50, penalty=None, eta0=0.1)
    -sgdreg.fit(x,y.ravel())
    -print(sgdreg.intercept_, sgdreg.coef_)
    +# To plot:
    +import matplotlib.pyplot as plt 
    +
    +
    +def f(x):
    +    return np.sin(2*np.pi*x + x**2)
    +
    +def f_grad_analytic(x):
    +    return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
    +
    +# Do the comparison:
    +x = np.linspace(0,1,1000)
    +
    +f_grad = egrad(f)
    +
    +computed = f_grad(x)
    +analytic = f_grad_analytic(x)
    +
    +plt.title('Derivative computed from Autograd compared with the analytical derivative')
    +plt.plot(x,computed,label='autograd')
    +plt.plot(x,analytic,label='analytic')
    +
    +plt.xlabel('x')
    +plt.ylabel('y')
    +plt.legend()
    +
    +plt.show()
    +
    +print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
     

    @@ -317,7 +351,7 @@ sgdreg.fit(x,y.

  • 52
  • 53
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs044.html b/doc/pub/Splines/html/._Splines-bs044.html index 6ea53a13c..327fc9132 100644 --- a/doc/pub/Splines/html/._Splines-bs044.html +++ b/doc/pub/Splines/html/._Splines-bs044.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,59 +273,35 @@ MathJax.Hub.Config({ -

    Gradient descent and Ridge

    +

    Using autograd

    -We have also discussed Ridge regression where the loss function contains a regularized given by the \( L_2 \) norm of \( \beta \), -$$ -C_{\text{ridge}}(\beta) = ||X\beta -\mathbf{y}||^2 + \lambda ||\beta||^2, \ \lambda \geq 0. -$$ - -

    -In order to minimize \( C_{\text{ridge}}(\beta) \) using GD we only have adjust the gradient as follows -$$ -\nabla_\beta C_{\text{ridge}}(\beta) = 2\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\ -\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\ -\end{bmatrix} + 2\lambda\begin{bmatrix} \beta_0 \\ \beta_1\end{bmatrix} = 2 (X^T(X\beta - \mathbf{y})+\lambda \beta). -$$ - -

    -We can now extend our program to minimize \( C_{\text{ridge}}(\beta) \) using gradient descent and compare with the analytical solution given by -$$ -\beta_{\text{ridge}} = \left(X^T X + \lambda I_{2 \times 2} \right)^{-1} X^T \mathbf{y}, -$$ - -for \( \lambda = {0,1,10,50,100} \) (\( \lambda = 0 \) corresponds to ordinary least squares). -We can then compute \( ||\beta_{\text{ridge}}|| \) for each \( \lambda \). +Here we +experiment with what kind of functions Autograd is capable +of finding the gradient of. The following Python functions are just +meant to illustrate what Autograd can do, but please feel free to +experiment with other, possibly more complicated, functions as well.

    -

    import numpy as np
    +
    import autograd.numpy as np
    +from autograd import grad
     
    -"""
    -The following setup is just a suggestion, feel free to write it the way you like.
    -"""
    +def f1(x):
    +    return x**3 + 1
     
    -#Setup problem described in the exercise
    -N  = 100 #Nr of datapoints
    -M  = 2   #Nr of features
    -x  = np.random.rand(N)
    -y  = 5*x**2 + 0.1*np.random.randn(N)
    +f1_grad = grad(f1)
     
    +# Remember to send in float as argument to the computed gradient from Autograd!
    +a = 1.0
     
    -#Compute analytic beta for Ridge regression 
    -X    = np.c_[np.ones(N),x]
    -XT_X = np.dot(X.T,X)
    +# See the evaluated gradient at a using autograd:
    +print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
     
    -l  = 0.1 #Ridge parameter lambda
    -Id = np.eye(XT_X.shape[0])
    -
    -Z = np.linalg.inv(XT_X+l*Id)
    -beta_ridge = np.dot(Z,np.dot(X.T,y))
    -
    -print(beta_ridge)
    -print(np.linalg.norm(beta_ridge)) #||beta||
    +# Compare with the analytical derivative, that is f1'(x) = 3*x**2 
    +grad_analytical = 3*a**2
    +print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
     

    @@ -350,7 +329,7 @@ beta_ridge = np

  • 53
  • 54
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs045.html b/doc/pub/Splines/html/._Splines-bs045.html index 490989853..e3d6ba6e7 100644 --- a/doc/pub/Splines/html/._Splines-bs045.html +++ b/doc/pub/Splines/html/._Splines-bs045.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,58 +273,53 @@ MathJax.Hub.Config({ -

    Automatic differentiation

    -Python has tools for so-called automatic differentiation. -Consider the following example -$$ -f(x) = \sin\left(2\pi x + x^2\right) -$$ +

    Autograd with more complicated functions

    -which has the following derivative -$$ -f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) -$$ - -Using autograd we have +

    +To differentiate with respect to two (or more) arguments of a Python +function, Autograd need to know at which variable the function if +being differentiated with respect to.

    import autograd.numpy as np
    +from autograd import grad
    +def f2(x1,x2):
    +    return 3*x1**3 + x2*(x1 - 5) + 1
     
    -# To do elementwise differentiation:
    -from autograd import elementwise_grad as egrad 
    +# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
    +f2_grad_x1 = grad(f2,0)
     
    -# To plot:
    -import matplotlib.pyplot as plt 
    +# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
    +f2_grad_x2 = grad(f2,1)
     
    +x1 = 1.0
    +x2 = 3.0 
     
    -def f(x):
    -    return np.sin(2*np.pi*x + x**2)
    +print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
    +print("-"*30)
     
    -def f_grad_analytic(x):
    -    return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
    +# Compare with the analytical derivatives:
     
    -# Do the comparison:
    -x = np.linspace(0,1,1000)
    +# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
    +f2_grad_x1_analytical = 9*x1**2 + x2
     
    -f_grad = egrad(f)
    +# Derivative of f2 w.r.t x2 is: x1 - 5:
    +f2_grad_x2_analytical = x1 - 5
     
    -computed = f_grad(x)
    -analytic = f_grad_analytic(x)
    +# See the evaluated derivations:
    +print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
    +print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
     
    -plt.title('Derivative computed from Autograd compared with the analytical derivative')
    -plt.plot(x,computed,label='autograd')
    -plt.plot(x,analytic,label='analytic')
    +print()
     
    -plt.xlabel('x')
    -plt.ylabel('y')
    -plt.legend()
    -
    -plt.show()
    -
    -print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
    +print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
    +print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
     
    +

    +Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable. +

    @@ -348,7 +346,7 @@ plt.show()

  • 54
  • 55
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs046.html b/doc/pub/Splines/html/._Splines-bs046.html index b16706cbc..650660d98 100644 --- a/doc/pub/Splines/html/._Splines-bs046.html +++ b/doc/pub/Splines/html/._Splines-bs046.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,38 +271,39 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Using autograd

    - -

    -Here we -experiment with what kind of functions Autograd is capable -of finding the gradient of. The following Python functions are just -meant to illustrate what Autograd can do, but please feel free to -experiment with other, possibly more complicated, functions as well. +

    More complicated functions using the elements of their arguments directly

    import autograd.numpy as np
     from autograd import grad
    +def f3(x): # Assumes x is an array of length 5 or higher
    +    return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
     
    -def f1(x):
    -    return x**3 + 1
    +f3_grad = grad(f3)
     
    -f1_grad = grad(f1)
    +x = np.linspace(0,4,5)
     
    -# Remember to send in float as argument to the computed gradient from Autograd!
    -a = 1.0
    +# Print the computed gradient:
    +print("The computed gradient of f3 is: ", f3_grad(x))
     
    -# See the evaluated gradient at a using autograd:
    -print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
    +# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
    +f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
     
    -# Compare with the analytical derivative, that is f1'(x) = 3*x**2 
    -grad_analytical = 3*a**2
    -print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
    +# Print the analytical gradient:
    +print("The analytical gradient of f3 is: ", f3_grad_analytical)
     
    +

    +Note that in this case, when sending an array as input argument, the +output from Autograd is another array. This is the true gradient of +the function, as opposed to the function in the previous example. By +using arrays to represent the variables, the output from Autograd +might be easier to work with, as the output is closer to what one +could expect form a gradient-evaluting function. +

    @@ -326,7 +330,7 @@ grad_analytical = 55

  • 56
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs047.html b/doc/pub/Splines/html/._Splines-bs047.html index 6d08d9546..1bbe71ede 100644 --- a/doc/pub/Splines/html/._Splines-bs047.html +++ b/doc/pub/Splines/html/._Splines-bs047.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,55 +271,31 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Autograd with more complicated functions

    - -

    -To differentiate with respect to two (or more) arguments of a Python -function, Autograd need to know at which variable the function if -being differentiated with respect to. +

    Functions using mathematical functions from Numpy

    import autograd.numpy as np
     from autograd import grad
    -def f2(x1,x2):
    -    return 3*x1**3 + x2*(x1 - 5) + 1
    +def f4(x):
    +    return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
     
    -# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
    -f2_grad_x1 = grad(f2,0)
    +f4_grad = grad(f4)
     
    -# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
    -f2_grad_x2 = grad(f2,1)
    +x = 2.7
     
    -x1 = 1.0
    -x2 = 3.0 
    +# Print the computed derivative:
    +print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
     
    -print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
    -print("-"*30)
    +# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
    +f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
     
    -# Compare with the analytical derivatives:
    -
    -# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
    -f2_grad_x1_analytical = 9*x1**2 + x2
    -
    -# Derivative of f2 w.r.t x2 is: x1 - 5:
    -f2_grad_x2_analytical = x1 - 5
    -
    -# See the evaluated derivations:
    -print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
    -print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
    -
    -print()
    -
    -print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
    -print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
    +# Print the analytical gradient:
    +print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
     
    -

    -Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable. -

    @@ -343,7 +322,7 @@ Note that the grad function will not produce the true gradient of the function.

  • 56
  • 57
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs048.html b/doc/pub/Splines/html/._Splines-bs048.html index 5627c0444..e833a627c 100644 --- a/doc/pub/Splines/html/._Splines-bs048.html +++ b/doc/pub/Splines/html/._Splines-bs048.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,37 +273,26 @@ MathJax.Hub.Config({ -

    More complicated functions using the elements of their arguments directly

    +

    More autograd

    import autograd.numpy as np
     from autograd import grad
    -def f3(x): # Assumes x is an array of length 5 or higher
    -    return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
    +def f5(x):
    +    if x >= 0:
    +        return x**2
    +    else:
    +        return -3*x + 1
     
    -f3_grad = grad(f3)
    +f5_grad = grad(f5)
     
    -x = np.linspace(0,4,5)
    +x = 2.7
     
    -# Print the computed gradient:
    -print("The computed gradient of f3 is: ", f3_grad(x))
    -
    -# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
    -f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
    -
    -# Print the analytical gradient:
    -print("The analytical gradient of f3 is: ", f3_grad_analytical)
    +# Print the computed derivative:
    +print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
     
    -

    -Note that in this case, when sending an array as input argument, the -output from Autograd is another array. This is the true gradient of -the function, as opposed to the function in the previous example. By -using arrays to represent the variables, the output from Autograd -might be easier to work with, as the output is closer to what one -could expect form a gradient-evaluting function. -

    @@ -327,7 +319,7 @@ could expect form a gradient-evaluting function.

  • 57
  • 58
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs049.html b/doc/pub/Splines/html/._Splines-bs049.html index 3170a1bee..61f865d3d 100644 --- a/doc/pub/Splines/html/._Splines-bs049.html +++ b/doc/pub/Splines/html/._Splines-bs049.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -268,30 +271,50 @@ MathJax.Hub.Config({

     

     

     

    - + -

    Functions using mathematical functions from Numpy

    +

    And with loops

    import autograd.numpy as np
     from autograd import grad
    -def f4(x):
    -    return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
    +def f6_for(x):
    +    val = 0
    +    for i in range(10):
    +        val = val + x**i
    +    return val
     
    -f4_grad = grad(f4)
    +def f6_while(x):
    +    val = 0
    +    i = 0
    +    while i < 10:
    +        val = val + x**i
    +        i = i + 1
    +    return val
     
    -x = 2.7
    +f6_for_grad = grad(f6_for)
    +f6_while_grad = grad(f6_while)
     
    -# Print the computed derivative:
    -print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
    +x = 0.5
     
    -# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
    -f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
    +# Print the computed derivaties of f6_for and f6_while
    +print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
    +print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
    +
    +

    -# Print the analytical gradient: -print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical)) + +

    import autograd.numpy as np
    +from autograd import grad
    +# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
    +# The analytical derivative is: sum(i*x**(i-1)) 
    +f6_grad_analytical = 0
    +for i in range(10):
    +    f6_grad_analytical += i*x**(i-1)
    +
    +print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
     

    @@ -319,7 +342,7 @@ f4_grad_analytical = x58

  • 59
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs050.html b/doc/pub/Splines/html/._Splines-bs050.html index f344ae0e0..9a31d590d 100644 --- a/doc/pub/Splines/html/._Splines-bs050.html +++ b/doc/pub/Splines/html/._Splines-bs050.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,26 +273,41 @@ MathJax.Hub.Config({ -

    More autograd

    - +

    Using recursion

    import autograd.numpy as np
     from autograd import grad
    -def f5(x):
    -    if x >= 0:
    -        return x**2
    +
    +def f7(n): # Assume that n is an integer
    +    if n == 1 or n == 0:
    +        return 1
         else:
    -        return -3*x + 1
    +        return n*f7(n-1)
     
    -f5_grad = grad(f5)
    +f7_grad = grad(f7)
     
    -x = 2.7
    +n = 2.0
     
    -# Print the computed derivative:
    -print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
    +print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
    +
    +# The function f7 is an implementation of the factorial of n.
    +# By using the product rule, one can find that the derivative is:
    +
    +f7_grad_analytical = 0
    +for i in range(int(n)-1):
    +    tmp = 1
    +    for k in range(int(n)-1):
    +        if k != i:
    +            tmp *= (n - k)
    +    f7_grad_analytical += tmp
    +
    +print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
     
    +

    +Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input. +

    @@ -316,7 +334,7 @@ x = 2.7

  • 59
  • 60
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs051.html b/doc/pub/Splines/html/._Splines-bs051.html index b798e64b1..546450928 100644 --- a/doc/pub/Splines/html/._Splines-bs051.html +++ b/doc/pub/Splines/html/._Splines-bs051.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,49 +273,29 @@ MathJax.Hub.Config({ -

    And with loops

    +

    Unsupported functions

    +Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd. +

    +Assigning a value to the variable being differentiated with respect to

    import autograd.numpy as np
     from autograd import grad
    -def f6_for(x):
    -    val = 0
    -    for i in range(10):
    -        val = val + x**i
    -    return val
    +def f8(x): # Assume x is an array
    +    x[2] = 3
    +    return x*2
     
    -def f6_while(x):
    -    val = 0
    -    i = 0
    -    while i < 10:
    -        val = val + x**i
    -        i = i + 1
    -    return val
    +f8_grad = grad(f8)
     
    -f6_for_grad = grad(f6_for)
    -f6_while_grad = grad(f6_while)
    +x = 8.4
     
    -x = 0.5
    -
    -# Print the computed derivaties of f6_for and f6_while
    -print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
    -print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
    +print("The derivative of f8 is:",f8_grad(x))
     

    +Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible. - -

    import autograd.numpy as np
    -from autograd import grad
    -# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
    -# The analytical derivative is: sum(i*x**(i-1)) 
    -f6_grad_analytical = 0
    -for i in range(10):
    -    f6_grad_analytical += i*x**(i-1)
    -
    -print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
    -

    @@ -339,7 +322,7 @@ f6_grad_analytical = 60

  • 61
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs052.html b/doc/pub/Splines/html/._Splines-bs052.html index c5b2a5204..4d939b2e5 100644 --- a/doc/pub/Splines/html/._Splines-bs052.html +++ b/doc/pub/Splines/html/._Splines-bs052.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,41 +273,45 @@ MathJax.Hub.Config({ -

    Using recursion

    +

    The syntax a.dot(b) when finding the dot product

    import autograd.numpy as np
     from autograd import grad
    +def f9(a): # Assume a is an array with 2 elements
    +    b = np.array([1.0,2.0])
    +    return a.dot(b)
     
    -def f7(n): # Assume that n is an integer
    -    if n == 1 or n == 0:
    -        return 1
    -    else:
    -        return n*f7(n-1)
    +f9_grad = grad(f9)
     
    -f7_grad = grad(f7)
    +x = np.array([1.0,0.0])
     
    -n = 2.0
    -
    -print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
    -
    -# The function f7 is an implementation of the factorial of n.
    -# By using the product rule, one can find that the derivative is:
    -
    -f7_grad_analytical = 0
    -for i in range(int(n)-1):
    -    tmp = 1
    -    for k in range(int(n)-1):
    -        if k != i:
    -            tmp *= (n - k)
    -    f7_grad_analytical += tmp
    -
    -print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
    +print("The derivative of f9 is:",f9_grad(x))
     

    -Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input. +Here we are told that the 'dot' function does not belong to Autograd's +version of a Numpy array. To overcome this, an alternative syntax +which also computed the dot product can be used: +

    + + +

    import autograd.numpy as np
    +from autograd import grad
    +def f9_alternative(x): # Assume a is an array with 2 elements
    +    b = np.array([1.0,2.0])
    +    return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
    +
    +f9_alternative_grad = grad(f9_alternative)
    +
    +x = np.array([3.0,0.0])
    +
    +print("The gradient of f9 is:",f9_alternative_grad(x))
    +
    +# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
    +# w.r.t x is (b_1, b_2).
    +

    @@ -331,7 +338,7 @@ Note that if n is equal to zero or one, Autograd will give an error message. Thi

  • 61
  • 62
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs053.html b/doc/pub/Splines/html/._Splines-bs053.html index bbb3292b3..6b90513cf 100644 --- a/doc/pub/Splines/html/._Splines-bs053.html +++ b/doc/pub/Splines/html/._Splines-bs053.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,29 +273,16 @@ MathJax.Hub.Config({ -

    Unsupported functions

    -Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd. - -

    -Assigning a value to the variable being differentiated with respect to +

    Recommended to avoid

    +The documentation recommends to avoid inplace operations such as

    -

    import autograd.numpy as np
    -from autograd import grad
    -def f8(x): # Assume x is an array
    -    x[2] = 3
    -    return x*2
    -
    -f8_grad = grad(f8)
    -
    -x = 8.4
    -
    -print("The derivative of f8 is:",f8_grad(x))
    +
    a += b
    +a -= b
    +a*= b
    +a /=b
     
    -

    -Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible. -

    @@ -319,7 +309,7 @@ Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The

  • 62
  • 63
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs054.html b/doc/pub/Splines/html/._Splines-bs054.html index 649b375a0..bf93aacab 100644 --- a/doc/pub/Splines/html/._Splines-bs054.html +++ b/doc/pub/Splines/html/._Splines-bs054.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,45 +273,39 @@ MathJax.Hub.Config({ -

    The syntax a.dot(b) when finding the dot product

    -

    - - -

    import autograd.numpy as np
    -from autograd import grad
    -def f9(a): # Assume a is an array with 2 elements
    -    b = np.array([1.0,2.0])
    -    return a.dot(b)
    -
    -f9_grad = grad(f9)
    -
    -x = np.array([1.0,0.0])
    -
    -print("The derivative of f9 is:",f9_grad(x))
    -
    -

    -Here we are told that the 'dot' function does not belong to Autograd's -version of a Numpy array. To overcome this, an alternative syntax -which also computed the dot product can be used: +

    Standard steepest descent

    +Before we proceed, we would like to discuss the approach called the +standard Steepest descent, which again leads to us having to be able +to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG). - -

    import autograd.numpy as np
    -from autograd import grad
    -def f9_alternative(x): # Assume a is an array with 2 elements
    -    b = np.array([1.0,2.0])
    -    return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
    +

    +The success of the CG method +for finding solutions of non-linear problems is based on the theory +of conjugate gradients for linear systems of equations. It belongs to +the class of iterative methods for solving problems from linear +algebra of the type +$$ +\begin{equation*} +\hat{A}\hat{x} = \hat{b}. +\end{equation*} +$$ -f9_alternative_grad = grad(f9_alternative) +

    +In the iterative process we end up with a problem like -x = np.array([3.0,0.0]) +$$ +\begin{equation*} + \hat{r}= \hat{b}-\hat{A}\hat{x}, +\end{equation*} +$$ -print("The gradient of f9 is:",f9_alternative_grad(x)) +where \( \hat{r} \) is the so-called residual or error in the iterative process. + +

    +When we have found the exact solution, \( \hat{r}=0 \). -# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively -# w.r.t x is (b_1, b_2). -

    @@ -335,7 +332,7 @@ x = np.a

  • 63
  • 64
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs055.html b/doc/pub/Splines/html/._Splines-bs055.html index 501de90bd..2b5a8f1cf 100644 --- a/doc/pub/Splines/html/._Splines-bs055.html +++ b/doc/pub/Splines/html/._Splines-bs055.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,16 +273,20 @@ MathJax.Hub.Config({ -

    Recommended to avoid

    -The documentation recommends to avoid inplace operations such as -

    +

    Gradient method

    + +

    +The residual is zero when we reach the minimum of the quadratic equation +$$ +\begin{equation*} + P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b}, +\end{equation*} +$$ + +

    +with the constraint that the matrix \( \hat{A} \) is positive definite and +symmetric. This defines also the Hessian and we want it to be positive definite. - -

    a += b
    -a -= b
    -a*= b
    -a /=b
    -

    @@ -306,7 +313,7 @@ a /=b

  • 64
  • 65
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs056.html b/doc/pub/Splines/html/._Splines-bs056.html index a5e5c302b..ddaf0383b 100644 --- a/doc/pub/Splines/html/._Splines-bs056.html +++ b/doc/pub/Splines/html/._Splines-bs056.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,21 +273,26 @@ MathJax.Hub.Config({ -

    Stochastic Gradient Descent

    +

    Steepest descent method

    -Stochastic gradient descent (SGD) and variants thereof address some of -the shortcomings of the Gradient descent method discussed above. +We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). +We can assume without loss of generality that +$$ +\begin{equation*} +\hat{x}_0=0, +\end{equation*} +$$ -

    -The underlying idea of SGD comes from the observation that the cost -function, which we want to minimize, can almost always be written as a -sum over \( n \) data points \( \{\mathbf{x}_i\}_{i=1}^n \), +or consider the system $$ -C(\mathbf{\beta}) = \sum_{i=1}^n c_i(\mathbf{x}_i, -\mathbf{\beta}). +\begin{equation*} +\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, +\end{equation*} $$ +instead. +

    @@ -311,7 +319,7 @@ $$

  • 65
  • 66
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs057.html b/doc/pub/Splines/html/._Splines-bs057.html index 5d74b562f..989d5af2f 100644 --- a/doc/pub/Splines/html/._Splines-bs057.html +++ b/doc/pub/Splines/html/._Splines-bs057.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,22 +273,33 @@ MathJax.Hub.Config({ -

    Computation of gradients

    - -

    -This in turn means that the gradient can be -computed as a sum over \( i \)-gradients +

    Steepest descent method

    +
    +
    +

    +One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form $$ -\nabla_\beta C(\mathbf{\beta}) = \sum_i^n \nabla_\beta c_i(\mathbf{x}_i, -\mathbf{\beta}). +\begin{equation*} + f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. +\end{equation*} $$ +This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition) +to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), +which equals +$$ +\begin{equation*} +\hat{A}\hat{x}_0-\hat{b}, +\end{equation*} +$$ + +and +\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). +

    -Stochasticity/randomness is introduced by only taking the -gradient on a subset of the data called minibatches. If there are \( n \) -data points and the size of each minibatch is \( M \), there will be \( n/M \) -minibatches. We denote these minibatches by \( B_k \) where -\( k=1,\cdots,n/M \). +

    +
    +

    @@ -313,7 +327,7 @@ minibatches. We denote these minibatches by \( B_k \) where

  • 66
  • 67
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs058.html b/doc/pub/Splines/html/._Splines-bs058.html index 92fb49673..412f8e49c 100644 --- a/doc/pub/Splines/html/._Splines-bs058.html +++ b/doc/pub/Splines/html/._Splines-bs058.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,27 +273,47 @@ MathJax.Hub.Config({ -

    SGD example

    -As an example, suppose we have \( 10 \) data points \( (\mathbf{x}_1,\cdots, \mathbf{x}_{10}) \) -and we choose to have \( M=5 \) minibathces, -then each minibatch contains two data points. In particular we have -\( B_1 = (\mathbf{x}_1,\mathbf{x}_2), \cdots, B_5 = -(\mathbf{x}_9,\mathbf{x}_{10}) \). Note that if you choose \( M=1 \) you -have only a single batch with all data points and on the other extreme, -you may choose \( M=n \) resulting in a minibatch for each datapoint, i.e -\( B_k = \mathbf{x}_k \). +

    Final expressions

    +
    +
    +

    +We can compute the residual iteratively as +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, + \end{equation*} +$$ -

    -The idea is now to approximate the gradient by replacing the sum over -all data points with a sum over the data points in one the minibatches -picked at random in each gradient descent step +which equals $$ -\nabla_{\beta} -C(\mathbf{\beta}) = \sum_{i=1}^n \nabla_\beta c_i(\mathbf{x}_i, -\mathbf{\beta}) \rightarrow \sum_{i \in B_k}^n \nabla_\beta -c_i(\mathbf{x}_i, \mathbf{\beta}). +\begin{equation*} +\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k), + \end{equation*} $$ +or +$$ +\begin{equation*} +(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k, + \end{equation*} +$$ + +which gives + +$$ +\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k} +$$ + +leading to the iterative scheme +$$ +\begin{equation*} +\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k}, + \end{equation*} +$$ +

    +
    + +

    @@ -317,7 +340,7 @@ $$

  • 67
  • 68
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs059.html b/doc/pub/Splines/html/._Splines-bs059.html index 26adbbbb5..eb8332702 100644 --- a/doc/pub/Splines/html/._Splines-bs059.html +++ b/doc/pub/Splines/html/._Splines-bs059.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,21 +273,7 @@ MathJax.Hub.Config({ -

    The gradient step

    - -

    -Thus a gradient descent step now looks like -$$ -\beta_{j+1} = \beta_j - \gamma_j \sum_{i \in B_k}^n \nabla_\beta c_i(\mathbf{x}_i, -\mathbf{\beta}) -$$ - -

    -where \( k \) is picked at random with equal -probability from \( [1,n/M] \). An iteration over the number of -minibathces (n/M) is commonly referred to as an epoch. Thus it is -typical to choose a number of epochs and for each epoch iterate over -the number of minibatches, as exemplified in the code below. +

    Code examples for steepest descent

    @@ -312,7 +301,7 @@ the number of minibatches, as exemplified in the code below.

  • 68
  • 69
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs060.html b/doc/pub/Splines/html/._Splines-bs060.html index b3709fd60..f2dcd6b51 100644 --- a/doc/pub/Splines/html/._Splines-bs060.html +++ b/doc/pub/Splines/html/._Splines-bs060.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,34 +273,43 @@ MathJax.Hub.Config({ -

    Simple example code

    - +

    Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come

    +
    +
    +

    - -

    import numpy as np 
    +
    +
    #include <cmath>
    +#include <iostream>
    +#include <fstream>
    +#include <iomanip>
    +#include "vectormatrixclass.h"
    +using namespace  std;
    +//   Main function begins here
    +int main(int  argc, char * argv[]){
    +  int dim = 2;
    +  Vector x(dim),xsd(dim), b(dim),x0(dim);
    +  Matrix A(dim,dim);
     
    -n = 100 #100 datapoints 
    -M = 5   #size of each minibatch
    -m = int(n/M) #number of minibatches
    -n_epochs = 10 #number of epochs
    -
    -j = 0
    -for epoch in range(1,n_epochs+1):
    -    for i in range(m):
    -        k = np.random.randint(m) #Pick the k-th minibatch at random
    -        #Compute the gradient using the data in minibatch Bk
    -        #Compute new suggestion for 
    -        j += 1
    +  // Set our initial guess
    +  x0(0) = x0(1) = 0;
    +  // Set the matrix
    +  A(0,0) =  3;    A(1,0) =  2;   A(0,1) =  2;   A(1,1) =  6;
    +  b(0) = 2; b(1) = -8;
    +  cout << "The Matrix A that we are using: " << endl;
    +  A.Print();
    +  cout << endl;
    +  xsd = SteepestDescent(A,b,x0);
    +  cout << "The approximate solution using Steepest Descent is: " << endl;
    +  xsd.Print();
    +  cout << endl;
    +}
     

    -Taking the gradient only on a subset of the data has two important -benefits. First, it introduces randomness which decreases the chance -that our opmization scheme gets stuck in a local minima. Second, if -the size of the minibatches are small relative to the number of -datapoints (\( M < n \)), the computation of the gradient is much -cheaper since we sum over the datapoints in the \( k-th \) minibatch and not -all \( n \) datapoints. +

    +
    +

    @@ -325,7 +337,7 @@ all \( n \) datapoints.

  • 69
  • 70
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs061.html b/doc/pub/Splines/html/._Splines-bs061.html index cb15dcc9c..138326cf5 100644 --- a/doc/pub/Splines/html/._Splines-bs061.html +++ b/doc/pub/Splines/html/._Splines-bs061.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,19 +273,39 @@ MathJax.Hub.Config({ -

    When do we stop?

    - +

    The routine for the steepest descent method

    +
    +
    +

    -A natural question is when do we stop the search for a new minimum? -One possibility is to compute the full gradient after a given number -of epochs and check if the norm of the gradient is smaller than some -threshold and stop if true. However, the condition that the gradient -is zero is valid also for local minima, so this would only tell us -that we are close to a local/global minimum. However, we could also -evaluate the cost function at this point, store the result and -continue the search. If the test kicks in at a later stage we can -compare the values of the cost function and keep the \( \beta \) that -gave the lowest value. + + +

    Vector SteepestDescent(Matrix A, Vector b, Vector x0){
    +  int IterMax, i;
    +  int dim = x0.Dimension();
    +  const double tolerance = 1.0e-14;
    +  Vector x(dim),f(dim),z(dim);
    +  double c,alpha,d;
    +  IterMax = 30;
    +  x = x0;
    +  r = A*x-b;
    +  i = 0;
    +  while (i <= IterMax){
    +    z = A*r;
    +    c = dot(r,r);
    +    alpha = c/dot(r,z);
    +    x = x - alpha*r;
    +    r =  A*x-b;
    +    if(sqrt(dot(r,r)) < tolerance) break;
    +    i++;
    +  }
    +  return x;
    +}
    +
    +

    +

    +
    +

    @@ -310,7 +333,7 @@ gave the lowest value.

  • 70
  • 71
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs062.html b/doc/pub/Splines/html/._Splines-bs062.html index 8efc67bc2..92b177ffb 100644 --- a/doc/pub/Splines/html/._Splines-bs062.html +++ b/doc/pub/Splines/html/._Splines-bs062.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,50 +273,71 @@ MathJax.Hub.Config({ -

    Slightly different approach

    - -

    -Another approach is to let the step length \( \gamma_j \) depend on the -number of epochs in such a way that it becomes very small after a -reasonable time such that we do not move at all. - -

    -As an example, let \( e = 0,1,2,3,\cdots \) denote the current epoch and let \( t_0, t_1 > 0 \) be two fixed numbers. Furthermore, let \( t = e \cdot m + i \) where \( m \) is the number of minibatches and \( i=0,\cdots,m-1 \). Then the function $$\gamma_j(t; t_0, t_1) = \frac{t_0}{t+t_1} $$ goes to zero as the number of epochs gets large. I.e. we start with a step length \( \gamma_j (0; t_0, t_1) = t_0/t_1 \) which decays in time \( t \). - -

    -In this way we can fix the number of epochs, compute \( \beta \) and -evaluate the cost function at the end. Repeating the computation will -give a different result since the scheme is random by design. Then we -pick the final \( \beta \) that gives the lowest value of the cost -function. +

    Steepest descent example

    -

    import numpy as np 
    +
    import numpy as np
    +import numpy.linalg as la
     
    -def step_length(t,t0,t1):
    -    return t0/(t+t1)
    +import scipy.optimize as sopt
     
    -n = 100 #100 datapoints 
    -M = 5   #size of each minibatch
    -m = int(n/M) #number of minibatches
    -n_epochs = 500 #number of epochs
    -t0 = 1.0
    -t1 = 10
    +import matplotlib.pyplot as pt
    +from mpl_toolkits.mplot3d import axes3d
     
    -gamma_j = t0/t1
    -j = 0
    -for epoch in range(1,n_epochs+1):
    -    for i in range(m):
    -        k = np.random.randint(m) #Pick the k-th minibatch at random
    -        #Compute the gradient using the data in minibatch Bk
    -        #Compute new suggestion for beta
    -        t = epoch*m+i
    -        gamma_j = step_length(t,t0,t1)
    -        j += 1
    +def f(x):
    +    return 0.5*x[0]**2 + 2.5*x[1]**2
     
    -print("gamma_j after %d epochs: %g" % (n_epochs,gamma_j))
    +def df(x):
    +    return np.array([x[0], 5*x[1]])
    +
    +fig = pt.figure()
    +ax = fig.gca(projection="3d")
    +
    +xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]
    +fmesh = f(np.array([xmesh, ymesh]))
    +ax.plot_surface(xmesh, ymesh, fmesh)
    +
    +

    +And then as countor plot +

    + + +

    pt.axis("equal")
    +pt.contour(xmesh, ymesh, fmesh)
    +guesses = [np.array([2, 2./5])]
    +
    +

    +Find guesses +

    + + +

    x = guesses[-1]
    +s = -df(x)
    +
    +

    +Run it! +

    + + +

    def f1d(alpha):
    +    return f(x + alpha*s)
    +
    +alpha_opt = sopt.golden(f1d)
    +next_guess = x + alpha_opt * s
    +guesses.append(next_guess)
    +print(next_guess)
    +
    +

    +What happened? +

    + + +

    pt.axis("equal")
    +pt.contour(xmesh, ymesh, fmesh, 50)
    +it_array = np.array(guesses)
    +pt.plot(it_array.T[0], it_array.T[1], "x-")
     

    @@ -340,6 +364,8 @@ j = 0

  • 70
  • 71
  • 72
  • +
  • ...
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs063.html b/doc/pub/Splines/html/._Splines-bs063.html index a80d1cda9..797231da1 100644 --- a/doc/pub/Splines/html/._Splines-bs063.html +++ b/doc/pub/Splines/html/._Splines-bs063.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,82 +273,34 @@ MathJax.Hub.Config({ -

    Program for stochastic gradient

    +

    Conjugate gradient method

    +
    +
    +

    +In the CG method we define so-called conjugate directions and two vectors +\( \hat{s} \) and \( \hat{t} \) +are said to be +conjugate if +$$ +\begin{equation*} +\hat{s}^T\hat{A}\hat{t}= 0. +\end{equation*} +$$ -

    +The philosophy of the CG method is to perform searches in various conjugate directions +of our vectors \( \hat{x}_i \) obeying the above criterion, namely +$$ +\begin{equation*} +\hat{x}_i^T\hat{A}\hat{x}_j= 0. +\end{equation*} +$$ - -

    # Importing various packages
    -from math import exp, sqrt
    -from random import random, seed
    -import numpy as np
    -import matplotlib.pyplot as plt
    -from sklearn.linear_model import SGDRegressor
    -
    -x = 2*np.random.rand(100,1)
    -y = 4+3*x+np.random.randn(100,1)
    -
    -xb = np.c_[np.ones((100,1)), x]
    -theta_linreg = np.linalg.inv(xb.T.dot(xb)).dot(xb.T).dot(y)
    -print("Own inversion")
    -print(theta_linreg)
    -sgdreg = SGDRegressor(n_iter = 50, penalty=None, eta0=0.1)
    -sgdreg.fit(x,y.ravel())
    -print("sgdreg from scikit")
    -print(sgdreg.intercept_, sgdreg.coef_)
    +Two vectors are conjugate if they are orthogonal with respect to 
    +this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \).
    +
    +
    -theta = np.random.randn(2,1) - -eta = 0.1 -Niterations = 1000 -m = 100 - -for iter in range(Niterations): - gradients = 2.0/m*xb.T.dot(xb.dot(theta)-y) - theta -= eta*gradients -print("theta frm own gd") -print(theta) - -xnew = np.array([[0],[2]]) -xbnew = np.c_[np.ones((2,1)), xnew] -ypredict = xbnew.dot(theta) -ypredict2 = xbnew.dot(theta_linreg) - - -n_epochs = 50 -t0, t1 = 5, 50 -m = 100 -def learning_schedule(t): - return t0/(t+t1) - -theta = np.random.randn(2,1) - -for epoch in range(n_epochs): - for i in range(m): - random_index = np.random.randint(m) - xi = xb[random_index:random_index+1] - yi = y[random_index:random_index+1] - gradients = 2 * xi.T.dot(xi.dot(theta)-yi) - eta = learning_schedule(epoch*m+i) - theta = theta - eta*gradients -print("theta from own sdg") -print(theta) - - - - - - -plt.plot(xnew, ypredict, "r-") -plt.plot(xnew, ypredict2, "b-") -plt.plot(x, y ,'ro') -plt.axis([0,2.0,0, 15.0]) -plt.xlabel(r'$x$') -plt.ylabel(r'$y$') -plt.title(r'Random numbers ') -plt.show() -

    @@ -370,6 +325,7 @@ plt.show()

  • 70
  • 71
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs064.html b/doc/pub/Splines/html/._Splines-bs064.html index cceb24815..eac493792 100644 --- a/doc/pub/Splines/html/._Splines-bs064.html +++ b/doc/pub/Splines/html/._Splines-bs064.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,17 +273,23 @@ MathJax.Hub.Config({ -

    Using gradient descent methods, limitations

    +

    Conjugate gradient method

    +
    +
    +

    +An example is given by the eigenvectors of the matrix +$$ +\begin{equation*} +\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j, +\end{equation*} +$$ -

      -
    • Gradient descent (GD) finds local minima of our function. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our energy function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.
    • -
    • GD is sensitive to initial conditions. One consequence of the local nature of GD is that initial conditions matter. Depending on where one starts, one will end up at a different local minima. Therefore, it is very important to think about how one initializes the training process. This is true for GD as well as more complicated variants of GD.
    • -
    • Gradients are computationally expensive to calculate for large datasets. In many cases in statistics and ML, the energy function is a sum of terms, with one term for each data point. For example, in linear regression, \( E \propto \sum_{i=1}^n (y_i - \mathbf{w}^T\cdot\mathbf{x}_i)^2 \); for logistic regression, the square error is replaced by the cross entropy. To calculate the gradient we have to sum over all \( n \) data points. Doing this at every GD step becomes extremely computationally expensive. An ingenious solution to this, is to calculate the gradients using small subsets of the data called "mini batches". This has the added benefit of introducing stochasticity into our algorithm.
    • -
    • GD is very sensitive to choices of learning rates. GD is extremely sensitive to the choice of learning rates. If the learning rate is very small, the training process take an extremely long time. For larger learning rates, GD can diverge and give poor results. Furthermore, depending on what the local landscape looks like, we have to modify the learning rates to ensure convergence. Ideally, we would adaptively choose the learning rates to match the landscape.
    • -
    • GD treats all directions in parameter space uniformly. Another major drawback of GD is that unlike Newton's method, the learning rate for GD is the same in all directions in parameter space. For this reason, the maximum learning rate is set by the behavior of the steepest direction and this can significantly slow down training. Ideally, we would like to take large steps in flat directions and small steps in steep directions. Since we are exploring rugged landscapes where curvatures change, this requires us to keep track of not only the gradient but second derivatives. The ideal scenario would be to calculate the Hessian but this proves to be too computationally expensive.
    • -
    • GD can take exponential time to escape saddle points, even with random initialization. As we mentioned, GD is extremely sensitive to initial condition since it determines the particular local minimum GD would eventually reach. However, even with a good initialization scheme, through the introduction of randomness, GD can still take exponential time to escape saddle points.
    • -
    +which is zero unless \( i=j \). +
    +
    + +

      @@ -303,6 +312,7 @@ MathJax.Hub.Config({
    • 70
    • 71
    • 72
    • +
    • 73
    • »
    diff --git a/doc/pub/Splines/html/._Splines-bs065.html b/doc/pub/Splines/html/._Splines-bs065.html index 326118108..663dfc650 100644 --- a/doc/pub/Splines/html/._Splines-bs065.html +++ b/doc/pub/Splines/html/._Splines-bs065.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,25 +273,30 @@ MathJax.Hub.Config({ -

    Momentum based GD

    - -

    -The stochastic gradient descent (SGD) is almost always used with a momentum or inertia term that serves as a memory of the direction we are moving in parameter space. This is typically -implemented as follows +

    Conjugate gradient method

    +
    +
    +

    +Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size +\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector $$ -\begin{align} -\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t) \nonumber \\ -\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t}, -\tag{2} -\end{align} +\begin{equation*} +\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}. +\end{equation*} $$ -where we have introduced a momentum parameter \( \gamma \), with \( 0\le\gamma\le 1 \), and for brevity we dropped the explicit notation to indicate the gradient is to be taken over a different mini-batch at each step. We call this algorithm gradient descent with momentum (GDM). From these equations, it is clear that \( \mathbf{v}_t \) is a running average of recently encountered gradients and \( (1-\gamma)^{-1} \) sets the characteristic time scale for the memory used in the averaging procedure. Consistent with this, when \( \gamma=0 \), this just reduces down to ordinary SGD as discussed earlier. An equivalent way of writing the updates is -$$ -\Delta \boldsymbol{\theta}_{t+1} = \gamma \Delta \boldsymbol{\theta}_t -\ \eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t), -$$ +We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions. +Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution +$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely + +$$ +\begin{equation*} + \hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i. +\end{equation*} +$$ +

    +
    -where we have defined \( \Delta \boldsymbol{\theta}_{t}= \boldsymbol{\theta}_t-\boldsymbol{\theta}_{t-1} \).

    @@ -312,6 +320,7 @@ where we have defined \( \Delta \boldsymbol{\theta}_{t}= \boldsymbol{\theta}_t-\

  • 70
  • 71
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs066.html b/doc/pub/Splines/html/._Splines-bs066.html index 633892f27..941d5524e 100644 --- a/doc/pub/Splines/html/._Splines-bs066.html +++ b/doc/pub/Splines/html/._Splines-bs066.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,24 +273,35 @@ MathJax.Hub.Config({ -

    More on momentum based approaches

    - -

    -Let us try to get more intuition from these equations. It is helpful to consider a simple physical analogy with a particle of mass \( m \) moving in a viscous medium with drag coefficient \( \mu \) and potential -\( E(\mathbf{w}) \). If we denote the particle's position by \( \mathbf{w} \), then its motion is described by +

    Conjugate gradient method

    +
    +
    +

    +The coefficients are given by $$ -m {d^2 \mathbf{w} \over dt^2} + \mu {d \mathbf{w} \over dt }= -\nabla_w E(\mathbf{w}). +\begin{equation*} + \mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. +\end{equation*} $$ -We can discretize this equation in the usual way to get +Multiplying with \( \hat{p}_k^T \) from the left gives + $$ -m { \mathbf{w}_{t+\Delta t}-2 \mathbf{w}_{t} +\mathbf{w}_{t-\Delta t} \over (\Delta t)^2}+\mu {\mathbf{w}_{t+\Delta t}- \mathbf{w}_{t} \over \Delta t} = -\nabla_w E(\mathbf{w}). +\begin{equation*} + \hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b}, +\end{equation*} $$ -Rearranging this equation, we can rewrite this as +and we can define the coefficients \( \alpha_k \) as + $$ -\Delta \mathbf{w}_{t +\Delta t}= - { (\Delta t)^2 \over m +\mu \Delta t} \nabla_w E(\mathbf{w})+ {m \over m +\mu \Delta t} \Delta \mathbf{w}_t. +\begin{equation*} + \alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k} +\end{equation*} $$ +

    +
    +

    @@ -310,6 +324,7 @@ $$

  • 70
  • 71
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs067.html b/doc/pub/Splines/html/._Splines-bs067.html index 0e4bb3071..3e7d0ec36 100644 --- a/doc/pub/Splines/html/._Splines-bs067.html +++ b/doc/pub/Splines/html/._Splines-bs067.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,32 +273,39 @@ MathJax.Hub.Config({ -

    Momentum parameter

    -Notice that this equation is identical to previous one if we identify the position of the particle, \( \mathbf{w} \), with the parameters \( \boldsymbol{\theta} \). This allows -us to identify the momentum parameter and learning rate with the mass of the particle and the viscous drag as: -$$ -\gamma= {m \over m +\mu \Delta t }, \qquad \eta = {(\Delta t)^2 \over m +\mu \Delta t}. -$$ - -Thus, as the name suggests, the momentum parameter is proportional to the mass of the particle and effectively provides inertia. Furthermore, in the large viscosity/small learning rate limit, our memory time scales as \( (1-\gamma)^{-1} \approx m/(\mu \Delta t) \). +

    Conjugate gradient method and iterations

    +
    +
    +

    -Why is momentum useful? SGD momentum helps the gradient descent algorithm gain speed in directions with persistent but small gradients even in the presence of stochasticity, while suppressing oscillations in high-curvature directions. This becomes especially important in situations where the landscape is shallow and flat in some directions and narrow and steep in others. It has been argued that first-order methods (with appropriate initial conditions) can perform comparable to more expensive second order methods, especially in the context of complex deep learning models. +If we choose the conjugate vectors \( \hat{p}_k \) carefully, +then we may not need all of them to obtain a good approximation to the solution +\( \hat{x} \). +We want to regard the conjugate gradient method as an iterative method. +This will us to solve systems where \( n \) is so large that the direct +method would take too much time.

    -These beneficial properties of momentum can sometimes become even more pronounced by using a slight modification of the classical momentum algorithm called Nesterov Accelerated Gradient (NAG). - -

    -In the NAG algorithm, rather than calculating the gradient at the current parameters, \( \nabla_\theta E(\boldsymbol{\theta}_t) \), one calculates the gradient at the expected value of the parameters given our current momentum, \( \nabla_\theta E(\boldsymbol{\theta}_t +\gamma \mathbf{v}_{t-1}) \). This yields the NAG update rule +We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). +We can assume without loss of generality that $$ -\begin{align} -\mathbf{v}_{t}&=\gamma \mathbf{v}_{t-1}+\eta_{t}\nabla_\theta E(\boldsymbol{\theta}_t +\gamma \mathbf{v}_{t-1}) \nonumber \\ -\boldsymbol{\theta}_{t+1}&= \boldsymbol{\theta}_t -\mathbf{v}_{t}. -\tag{3} -\end{align} +\begin{equation*} +\hat{x}_0=0, +\end{equation*} $$ -One of the major advantages of NAG is that it allows for the use of a larger learning rate than GDM for the same choice of \( \gamma \). +or consider the system +$$ +\begin{equation*} +\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, +\end{equation*} +$$ + +instead. +

    +
    +

    @@ -317,6 +327,7 @@ One of the major advantages of NAG is that it allows for the use of a larger lea

  • 70
  • 71
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs068.html b/doc/pub/Splines/html/._Splines-bs068.html index 73a623360..c41d545aa 100644 --- a/doc/pub/Splines/html/._Splines-bs068.html +++ b/doc/pub/Splines/html/._Splines-bs068.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,27 +273,33 @@ MathJax.Hub.Config({ -

    Second moment of the gradient

    +

    Conjugate gradient method

    +
    +
    +

    +One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form +$$ +\begin{equation*} + f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. +\end{equation*} +$$ -

    -In stochastic gradient descent, with and without momentum, we still -have to specify a schedule for tuning the learning rates \( \eta_t \) -as a function of time. As discussed in the context of Newton's -method, this presents a number of dilemmas. The learning rate is -limited by the steepest direction which can change depending on the -current position in the landscape. To circumvent this problem, ideally -our algorithm would keep track of curvature and take large steps in -shallow, flat directions and small steps in steep, narrow directions. -Second-order methods accomplish this by calculating or approximating -the Hessian and normalizing the learning rate by the -curvature. However, this is very computationally expensive for -extremely large models. Ideally, we would like to be able to -adaptively change the step size to match the landscape without paying -the steep computational price of calculating or approximating -Hessians. +This suggests taking the first basis vector \( \hat{p}_1 \) +to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), +which equals +$$ +\begin{equation*} +\hat{A}\hat{x}_0-\hat{b}, +\end{equation*} +$$ + +and +\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). +The other vectors in the basis will be conjugate to the gradient, +hence the name conjugate gradient method. +

    +
    -

    -Recently, a number of methods have been introduced that accomplish this by tracking not only the gradient, but also the second moment of the gradient. These methods include AdaGrad, AdaDelta, RMS-Prop, and ADAM.

    @@ -311,6 +320,7 @@ Recently, a number of methods have been introduced that accomplish this by track

  • 70
  • 71
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs069.html b/doc/pub/Splines/html/._Splines-bs069.html index aa507ccd0..6e47614c0 100644 --- a/doc/pub/Splines/html/._Splines-bs069.html +++ b/doc/pub/Splines/html/._Splines-bs069.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,20 +273,32 @@ MathJax.Hub.Config({ -

    RMS prop

    - -

    -In RMS prop, in addition to keeping a running average of the first moment of the gradient, we also keep track of the second moment denoted by \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}_t^2] \). The update rule for RMS prop is given by +

    Conjugate gradient method

    +
    +
    +

    +Let \( \hat{r}_k \) be the residual at the \( k \)-th step: $$ -\begin{align} -\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta}) -\tag{4}\\ -\mathbf{s}_t &=\beta \mathbf{s}_{t-1} +(1-\beta)\mathbf{g}_t^2 \nonumber \\ -\boldsymbol{\theta}_{t+1}&=&\boldsymbol{\theta}_t - \eta_t { \mathbf{g}_t \over \sqrt{\mathbf{s}_t +\epsilon}}, \nonumber -\end{align} +\begin{equation*} +\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k. +\end{equation*} $$ -where \( \beta \) controls the averaging time of the second moment and is typically taken to be about \( \beta=0.9 \), \( \eta_t \) is a learning rate typically chosen to be \( 10^{-3} \), and \( \epsilon\sim 10^{-8} \) is a small regularization constant to prevent divergences. Multiplication and division by vectors is understood as an element-wise operation. It is clear from this formula that the learning rate is reduced in directions where the norm of the gradient is consistently large. This greatly speeds up the convergence by allowing us to use a larger learning rate for flat directions. +Note that \( \hat{r}_k \) is the negative gradient of \( f \) at +\( \hat{x}=\hat{x}_k \), +so the gradient descent method would be to move in the direction \( \hat{r}_k \). +Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other, +so we take the direction closest to the gradient \( \hat{r}_k \) +under the conjugacy constraint. +This gives the following expression +$$ +\begin{equation*} +\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k. +\end{equation*} +$$ +

    +
    +

    @@ -303,6 +318,7 @@ where \( \beta \) controls the averaging time of the second moment and is typica

  • 70
  • 71
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs070.html b/doc/pub/Splines/html/._Splines-bs070.html index 84265e551..e74b519cf 100644 --- a/doc/pub/Splines/html/._Splines-bs070.html +++ b/doc/pub/Splines/html/._Splines-bs070.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,31 +273,42 @@ MathJax.Hub.Config({ -

    ADAM optimizer

    - -

    -A related algorithm is the ADAM optimizer. In ADAM, we keep a running average of both the first and second moment of the gradient and use this information to adaptively change the learning rate for different parameters. In addition to keeping a running average of the first and second moments of the gradient (i.e. \( \mathbf{m}_t=\mathbb{E}[\mathbf{g}_t] \) and \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}^2_t] \), respectively), ADAM performs an additional bias correction to account for the fact that we are estimating the first two moments of the gradient using a running average (denoted by the hats in the update rule below). The update rule for ADAM is given by (where multiplication and division are once again understood to be element-wise operations below) +

    Conjugate gradient method

    +
    +
    +

    +We can also compute the residual iteratively as $$ -\begin{align} -\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta}) -\tag{5}\\ -\mathbf{m}_t &= \beta_1 \mathbf{m}_{t-1} + (1-\beta_1) \mathbf{g}_t \nonumber \\ -\mathbf{s}_t &=\beta_2 \mathbf{s}_{t-1} +(1-\beta_2)\mathbf{g}_t^2 \nonumber \\ -\hat{\mathbf{m}}_t&={\mathbf{m}_t \over 1-\beta_1^t} \nonumber \\ -\hat{\mathbf{s}}_t &={\mathbf{s}_t \over1-\beta_2^t} \nonumber \\ -\boldsymbol{\theta}_{t+1}&=\boldsymbol{\theta}_t - \eta_t { \hat{\mathbf{m}}_t \over \sqrt{\hat{\mathbf{s}}_t} +\epsilon}, \nonumber \\ -\tag{6} -\end{align} +\begin{equation*} +\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, + \end{equation*} $$ -where \( \beta_1 \) and \( \beta_2 \) set the memory lifetime of the first and second moment and are typically taken to be \( 0.9 \) and \( 0.99 \) respectively, and \( \eta \) and \( \epsilon \) are identical to RMSprop. +which equals +$$ +\begin{equation*} +\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k), + \end{equation*} +$$ -

    -Like in RMSprop, the effective step size of a parameter depends on the magnitude of its gradient squared. To understand this better, let us rewrite this expression in terms of the variance \( \boldsymbol{\sigma}_t^2 = \hat{\mathbf{s}}_t - (\hat{\mathbf{m}}_t)^2 \). Consider a single parameter \( \theta_t \). The update rule for this parameter is given by +or $$ -\Delta \theta_{t+1}= -\eta_t { \hat{m}_t \over \sqrt{\sigma_t^2 + m_t^2 }+\epsilon}. +\begin{equation*} +(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k, + \end{equation*} $$ +which gives + +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k}, + \end{equation*} +$$ +

    +
    + +

    @@ -312,6 +326,7 @@ $$

  • 70
  • 71
  • 72
  • +
  • 73
  • »
  • diff --git a/doc/pub/Splines/html/._Splines-bs071.html b/doc/pub/Splines/html/._Splines-bs071.html index bfcaad2b2..a072cecc1 100644 --- a/doc/pub/Splines/html/._Splines-bs071.html +++ b/doc/pub/Splines/html/._Splines-bs071.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -270,17 +273,44 @@ MathJax.Hub.Config({ -

    Practical tips

    +

    Simple implementation of the Conjugate gradient algorithm

    +
    +
    +

    +

    -

      -
    • Randomize the data when making mini-batches. It is always important to randomly shuffle the data when forming mini-batches. Otherwise, the gradient descent method can fit spurious correlations resulting from the order in which data is presented.
    • -
    • Transform your inputs. Learning becomes difficult when our landscape has a mixture of steep and flat directions. One simple trick for minimizing these situations is to standardize the data by subtracting the mean and normalizing the variance of input variables. Whenever possible, also decorrelate the inputs. To understand why this is helpful, consider the case of linear regression. It is easy to show that for the squared error cost function, the Hessian of the energy matrix is just the correlation matrix between the inputs. Thus, by standardizing the inputs, we are ensuring that the landscape looks homogeneous in all directions in parameter space. Since most deep networks can be viewed as linear transformations followed by a non-linearity at each layer, we expect this intuition to hold beyond the linear case.
    • -
    • Monitor the out-of-sample performance. Always monitor the performance of your model on a validation set (a small portion of the training data that is held out of the training process to serve as a proxy for the test set. If the validation error starts increasing, then the model is beginning to overfit. Terminate the learning process. This early stopping significantly improves performance in many settings.
    • -
    • Adaptive optimization methods don't always have good generalization. Recent studies have shown that adaptive methods such as ADAM, RMSPorp, and AdaGrad tend to have poor generalization compared to SGD or SGD with momentum, particularly in the high-dimensional limit (i.e. the number of parameters exceeds the number of data points). Although it is not clear at this stage why these methods perform so well in training deep neural networks, simpler procedures like properly-tuned SGD may work as well or better in these applications.
    • -
    + +
      Vector ConjugateGradient(Matrix A, Vector b, Vector x0){
    +  int dim = x0.Dimension();
    +  const double tolerance = 1.0e-14;
    +  Vector x(dim),r(dim),v(dim),z(dim);
    +  double c,t,d;
     
    -Geron's text, see chapter 11, has several interesting discussions.
    +  x = x0;
    +  r = b - A*x;
    +  v = r;
    +  c = dot(r,r);
    +  int i = 0; IterMax = dim;
    +  while(i <= IterMax){
    +    z = A*v;
    +    t = c/dot(v,z);
    +    x = x + t*v;
    +    r = r - t*z;
    +    d = dot(r,r);
    +    if(sqrt(d) < tolerance)
    +      break;
    +    v = r + (d/c)*v;
    +    c = d;  i++;
    +  }
    +  return x;
    +} 
    +
    +

    +

    +
    + +

      @@ -296,6 +326,8 @@ Geron's text, see chapter 11, has several interesting discussions.
    • 70
    • 71
    • 72
    • +
    • 73
    • +
    • »
    diff --git a/doc/pub/Splines/html/._Splines-bs072.html b/doc/pub/Splines/html/._Splines-bs072.html index 7921ef55f..38a0d2219 100644 --- a/doc/pub/Splines/html/._Splines-bs072.html +++ b/doc/pub/Splines/html/._Splines-bs072.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,115 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -186,79 +186,78 @@ MathJax.Hub.Config({ @@ -274,32 +273,41 @@ MathJax.Hub.Config({ -

    ADAM optimizer

    +

    Broyden–Fletcher–Goldfarb–Shanno algorithm

    +
    +
    +

    +The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take.

    -A related algorithm is the ADAM optimizer. In ADAM, we keep a running average of both the first and second moment of the gradient and use this information to adaptively change the learning rate for different parameters. In addition to keeping a running average of the first and second moments of the gradient (i.e. \( \mathbf{m}_t=\mathbb{E}[\mathbf{g}_t] \) and \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}^2_t] \), respectively), ADAM performs an additional bias correction to account for the fact that we are estimating the first two moments of the gradient using a running average (denoted by the hats in the update rule below). The update rule for ADAM is given by (where multiplication and division are once again understood to be element-wise operations below) -$$ -\begin{align} -\mathbf{g}_t &= \nabla_\theta E(\boldsymbol{\theta}) -\tag{5}\\ -\mathbf{m}_t &= \beta_1 \mathbf{m}_{t-1} + (1-\beta_1) \mathbf{g}_t \nonumber \\ -\mathbf{s}_t &=\beta_2 \mathbf{s}_{t-1} +(1-\beta_2)\mathbf{g}_t^2 \nonumber \\ -\hat{\mathbf{m}}_t&={\mathbf{m}_t \over 1-\beta_1^t} \nonumber \\ -\hat{\mathbf{s}}_t &={\mathbf{s}_t \over1-\beta_2^t} \nonumber \\ -\boldsymbol{\theta}_{t+1}&=\boldsymbol{\theta}_t - \eta_t { \hat{\mathbf{m}}_t \over \sqrt{\hat{\mathbf{s}}_t} +\epsilon}, \nonumber \\ -\tag{6} -\end{align} -$$ - -where \( \beta_1 \) and \( \beta_2 \) set the memory lifetime of the first and second moment and are typically taken to be \( 0.9 \) and \( 0.99 \) respectively, and \( \eta \) and \( \epsilon \) are identical to RMSprop. +The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage.

    -Like in RMSprop, the effective step size of a parameter depends on the magnitude of its gradient squared. To understand this better, let us rewrite this expression in terms of the variance \( \boldsymbol{\sigma}_t^2 = \hat{\mathbf{s}}_t - (\hat{\mathbf{m}}_t)^2 \). Consider a single parameter \( \theta_t \). The update rule for this parameter is given by +The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation $$ -\Delta \theta_{t+1}= -\eta_t { \hat{m}_t \over \sqrt{\sigma_t^2 + m_t^2 }+\epsilon}. +B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}), $$

    +where \( B_{k} \) is an approximation to the Hessian matrix, which is +updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \) +is the gradient of the function +evaluated at \( x_k \). +A line search in the direction \( p_k \) is then used to +find the next point \( x_{k+1} \) by minimising +$$ +f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}), +$$ + +over the scalar \( \alpha > 0 \). + +

    +

    +
    + + +

    +

    diff --git a/doc/pub/Splines/html/Splines-bs.html b/doc/pub/Splines/html/Splines-bs.html index 9941cf5dd..b905e18e2 100644 --- a/doc/pub/Splines/html/Splines-bs.html +++ b/doc/pub/Splines/html/Splines-bs.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -40,113 +41,114 @@ Automatically generated HTML file from DocOnce source @@ -184,77 +186,78 @@ MathJax.Hub.Config({ @@ -289,7 +292,7 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 18, 2018

    +

    Sep 19, 2019


    @@ -313,7 +316,7 @@ MathJax.Hub.Config({

  • 9
  • 10
  • ...
  • -
  • 72
  • +
  • 73
  • »
  • @@ -331,7 +334,7 @@ MathJax.Hub.Config({
    - © 1999-2018, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license + © 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
    diff --git a/doc/pub/Splines/html/Splines-reveal.html b/doc/pub/Splines/html/Splines-reveal.html index e3efed135..85d3f9ad3 100644 --- a/doc/pub/Splines/html/Splines-reveal.html +++ b/doc/pub/Splines/html/Splines-reveal.html @@ -1,8 +1,8 @@ -\ + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -148,18 +148,23 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

     
    -

    Oct 18, 2018

    +

    Sep 19, 2019


    - © 1999-2018, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license + © 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
    -

    Optimization, the central part of any Machine Learning algortithm

    +

    Optimization problems, why?

    +
    + + +
    +

    Optimization, the central part of any Machine Learning algortithm

    Almost every problem in machine learning and data science starts with @@ -174,7 +179,7 @@ some approximative/numerical method to compute the minimum.

    -

    Revisiting our Logistic Regression case

    +

    Revisiting our Logistic Regression case

    In our discussion on Logistic Regression we studied the @@ -198,7 +203,7 @@ where \( \hat{\beta} \) are the weights we wish to extract from data, in our cas

    -

    The equations to solve

    +

    The equations to solve

    Our compact equations used a definition of a vector \( \hat{y} \) with \( n \) @@ -228,7 +233,7 @@ This defines what is called the Hessian matrix.

    -

    Solving using Newton-Raphson's method

    +

    Solving using Newton-Raphson's method

    If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives. @@ -258,7 +263,7 @@ If we can compute these matrices, in particular the Hessian, the above is often

    -

    Brief reminder on Newton-Raphson's method

    +

    Brief reminder on Newton-Raphson's method

    Let us quickly remind ourselves how we derive the above method. @@ -275,7 +280,7 @@ normally discourage the use of this method.

    -

    The equations

    +

    The equations

    The Newton-Raphson formula consists geometrically of extending the @@ -319,7 +324,7 @@ $$

    -

    Simple geometric interpretation

    +

    Simple geometric interpretation

    The above is Newton-Raphson's method. It has a simple geometric @@ -337,7 +342,7 @@ vanishes, then Newton-Raphson may fail totally

    -

    Extending to more than one variable

    +

    Extending to more than one variable

    Newton's method can be generalized to systems of several non-linear equations @@ -402,7 +407,7 @@ more than two non-linear equations. In our case, the Jacobian matrix is given by

    -

    Steepest descent

    +

    Steepest descent

    The basic idea of gradient descent is @@ -428,7 +433,7 @@ we are always moving towards smaller function values, i.e a minimum.

    -

    More on Steepest descent

    +

    More on Steepest descent

    The previous observation is the basis of the method of steepest @@ -449,7 +454,7 @@ the learning rate within the context of Machine Learning.

    -

    The ideal

    +

    The ideal

    Ideally the sequence \( \{\mathbf{x}_k \}_{k=0} \) converges to a global @@ -475,7 +480,7 @@ Note that the gradient is a function of \( \mathbf{x} =

    -

    The sensitiveness of the gradient descent

    +

    The sensitiveness of the gradient descent

    The gradient descent method @@ -494,7 +499,7 @@ randomness. One such method is that of Stochastic Gradient Descent

    -

    Convex functions

    +

    Convex functions

    Ideally we want our cost/loss function to be convex(concave). @@ -514,7 +519,7 @@ regular polygons (triangles, rectangles, pentagons, etc...).

    -

    Convex function

    +

    Convex function

    Convex function: Let \( X \subset \mathbb{R}^n \) be a convex set. Assume that the function \( f: X \rightarrow \mathbb{R} \) is continuous, then \( f \) is said to be convex if

     
    @@ -524,7 +529,7 @@ $$f(tx_1 + (1-t)x_2) \leq tf(x_1) + (1-t)f(x_2) $$

    -

    Conditions on convex functions

    +

    Conditions on convex functions

    In the following we state first and second-order conditions which @@ -566,7 +571,7 @@ This condition is particularly useful since it gives us an procedure for determi

    -

    More on convex functions

    +

    More on convex functions

    The next result is of great importance to us and the reason why we are @@ -595,7 +600,7 @@ This result means that if we know that the cost/loss function is convex and we a

    -

    Some simple problems

    +

    Some simple problems

    1. Show that \( f(x)=x^2 \) is convex for \( x \in \mathbb{R} \) using the definition of convexity. Hint: If you re-write the definition, \( f \) is convex if the following holds for all \( x,y \in D_f \) and any \( \lambda \in [0,1] \) $\lambda f(x)+(1-\lambda)f(y)-f(\lambda x + (1-\lambda) y ) \geq 0$.
    2. @@ -622,666 +627,7 @@ Using the definition of convexity, try to show that a function satisfying the pr
      -

      Standard steepest descent

      - -

      -Before we proceed, we would like to discuss the approach called the -standard Steepest descent, which again leads to us having to be able -to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG). - -

      -The success of the CG method -for finding solutions of non-linear problems is based on the theory -of conjugate gradients for linear systems of equations. It belongs to -the class of iterative methods for solving problems from linear -algebra of the type -

       
      -$$ -\begin{equation*} -\hat{A}\hat{x} = \hat{b}. -\end{equation*} -$$ -

       
      - -

      -In the iterative process we end up with a problem like - -

       
      -$$ -\begin{equation*} - \hat{r}= \hat{b}-\hat{A}\hat{x}, -\end{equation*} -$$ -

       
      - -where \( \hat{r} \) is the so-called residual or error in the iterative process. - -

      -When we have found the exact solution, \( \hat{r}=0 \). -

      - - -
      -

      Gradient method

      - -

      -The residual is zero when we reach the minimum of the quadratic equation -

       
      -$$ -\begin{equation*} - P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b}, -\end{equation*} -$$ -

       
      - -

      -with the constraint that the matrix \( \hat{A} \) is positive definite and -symmetric. This defines also the Hessian and we want it to be positive definite. -

      - - -
      -

      Steepest descent method

      - -

      -We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). -We can assume without loss of generality that -

       
      -$$ -\begin{equation*} -\hat{x}_0=0, -\end{equation*} -$$ -

       
      - -or consider the system -

       
      -$$ -\begin{equation*} -\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, -\end{equation*} -$$ -

       
      - -instead. -

      - - -
      -

      Steepest descent method

      -
      - -

      -One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form -

       
      -$$ -\begin{equation*} - f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. -\end{equation*} -$$ -

       
      - -This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition) -to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), -which equals -

       
      -$$ -\begin{equation*} -\hat{A}\hat{x}_0-\hat{b}, -\end{equation*} -$$ -

       
      - -and -\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). - - -

      -
      - - -
      -

      Final expressions

      -
      - -

      -We can compute the residual iteratively as -

       
      -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, - \end{equation*} -$$ -

       
      - -which equals -

       
      -$$ -\begin{equation*} -\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k), - \end{equation*} -$$ -

       
      - -or -

       
      -$$ -\begin{equation*} -(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k, - \end{equation*} -$$ -

       
      - -which gives - -

       
      -$$ -\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k} -$$ -

       
      - -leading to the iterative scheme -

       
      -$$ -\begin{equation*} -\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k}, - \end{equation*} -$$ -

       
      -

      -
      - - -
      -

      Code examples for steepest descent

      -
      - - -
      -

      Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come

      -
      - -

      - - -

      #include <cmath>
      -#include <iostream>
      -#include <fstream>
      -#include <iomanip>
      -#include "vectormatrixclass.h"
      -using namespace  std;
      -//   Main function begins here
      -int main(int  argc, char * argv[]){
      -  int dim = 2;
      -  Vector x(dim),xsd(dim), b(dim),x0(dim);
      -  Matrix A(dim,dim);
      -
      -  // Set our initial guess
      -  x0(0) = x0(1) = 0;
      -  // Set the matrix
      -  A(0,0) =  3;    A(1,0) =  2;   A(0,1) =  2;   A(1,1) =  6;
      -  b(0) = 2; b(1) = -8;
      -  cout << "The Matrix A that we are using: " << endl;
      -  A.Print();
      -  cout << endl;
      -  xsd = SteepestDescent(A,b,x0);
      -  cout << "The approximate solution using Steepest Descent is: " << endl;
      -  xsd.Print();
      -  cout << endl;
      -}
      -
      - -
      -
      - - -
      -

      The routine for the steepest descent method

      -
      - -

      - - -

      Vector SteepestDescent(Matrix A, Vector b, Vector x0){
      -  int IterMax, i;
      -  int dim = x0.Dimension();
      -  const double tolerance = 1.0e-14;
      -  Vector x(dim),f(dim),z(dim);
      -  double c,alpha,d;
      -  IterMax = 30;
      -  x = x0;
      -  r = A*x-b;
      -  i = 0;
      -  while (i <= IterMax){
      -    z = A*r;
      -    c = dot(r,r);
      -    alpha = c/dot(r,z);
      -    x = x - alpha*r;
      -    r =  A*x-b;
      -    if(sqrt(dot(r,r)) < tolerance) break;
      -    i++;
      -  }
      -  return x;
      -}
      -
      - -
      -
      - - -
      -

      Steepest descent example

      - -

      - - -

      import numpy as np
      -import numpy.linalg as la
      -
      -import scipy.optimize as sopt
      -
      -import matplotlib.pyplot as pt
      -from mpl_toolkits.mplot3d import axes3d
      -
      -def f(x):
      -    return 0.5*x[0]**2 + 2.5*x[1]**2
      -
      -def df(x):
      -    return np.array([x[0], 5*x[1]])
      -
      -fig = pt.figure()
      -ax = fig.gca(projection="3d")
      -
      -xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]
      -fmesh = f(np.array([xmesh, ymesh]))
      -ax.plot_surface(xmesh, ymesh, fmesh)
      -
      -

      -And then as countor plot -

      - - -

      pt.axis("equal")
      -pt.contour(xmesh, ymesh, fmesh)
      -guesses = [np.array([2, 2./5])]
      -
      -

      -Find guesses -

      - - -

      x = guesses[-1]
      -s = -df(x)
      -
      -

      -Run it! -

      - - -

      def f1d(alpha):
      -    return f(x + alpha*s)
      -
      -alpha_opt = sopt.golden(f1d)
      -next_guess = x + alpha_opt * s
      -guesses.append(next_guess)
      -print(next_guess)
      -
      -

      -What happened? -

      - - -

      pt.axis("equal")
      -pt.contour(xmesh, ymesh, fmesh, 50)
      -it_array = np.array(guesses)
      -pt.plot(it_array.T[0], it_array.T[1], "x-")
      -
      -
      - - -
      -

      Conjugate gradient method

      -
      - -

      -In the CG method we define so-called conjugate directions and two vectors -\( \hat{s} \) and \( \hat{t} \) -are said to be -conjugate if -

       
      -$$ -\begin{equation*} -\hat{s}^T\hat{A}\hat{t}= 0. -\end{equation*} -$$ -

       
      - -The philosophy of the CG method is to perform searches in various conjugate directions -of our vectors \( \hat{x}_i \) obeying the above criterion, namely -

       
      -$$ -\begin{equation*} -\hat{x}_i^T\hat{A}\hat{x}_j= 0. -\end{equation*} -$$ -

       
      - -Two vectors are conjugate if they are orthogonal with respect to -this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \). -

      -
      - - -
      -

      Conjugate gradient method

      -
      - -

      -An example is given by the eigenvectors of the matrix -

       
      -$$ -\begin{equation*} -\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j, -\end{equation*} -$$ -

       
      - -which is zero unless \( i=j \). -

      -
      - - -
      -

      Conjugate gradient method

      -
      - -

      -Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size -\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector -

       
      -$$ -\begin{equation*} -\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}. -\end{equation*} -$$ -

       
      - -We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions. -Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution -$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely - -

       
      -$$ -\begin{equation*} - \hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i. -\end{equation*} -$$ -

       
      -

      -
      - - -
      -

      Conjugate gradient method

      -
      - -

      -The coefficients are given by -

       
      -$$ -\begin{equation*} - \mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. -\end{equation*} -$$ -

       
      - -Multiplying with \( \hat{p}_k^T \) from the left gives - -

       
      -$$ -\begin{equation*} - \hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b}, -\end{equation*} -$$ -

       
      - -and we can define the coefficients \( \alpha_k \) as - -

       
      -$$ -\begin{equation*} - \alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k} -\end{equation*} -$$ -

       
      -

      -
      - - -
      -

      Conjugate gradient method and iterations

      -
      - -

      -If we choose the conjugate vectors \( \hat{p}_k \) carefully, -then we may not need all of them to obtain a good approximation to the solution -\( \hat{x} \). -We want to regard the conjugate gradient method as an iterative method. -This will us to solve systems where \( n \) is so large that the direct -method would take too much time. - -

      -We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). -We can assume without loss of generality that -

       
      -$$ -\begin{equation*} -\hat{x}_0=0, -\end{equation*} -$$ -

       
      - -or consider the system -

       
      -$$ -\begin{equation*} -\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, -\end{equation*} -$$ -

       
      - -instead. -

      -
      - - -
      -

      Conjugate gradient method

      -
      - -

      -One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form -

       
      -$$ -\begin{equation*} - f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. -\end{equation*} -$$ -

       
      - -This suggests taking the first basis vector \( \hat{p}_1 \) -to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), -which equals -

       
      -$$ -\begin{equation*} -\hat{A}\hat{x}_0-\hat{b}, -\end{equation*} -$$ -

       
      - -and -\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). -The other vectors in the basis will be conjugate to the gradient, -hence the name conjugate gradient method. -

      -
      - - -
      -

      Conjugate gradient method

      -
      - -

      -Let \( \hat{r}_k \) be the residual at the \( k \)-th step: -

       
      -$$ -\begin{equation*} -\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k. -\end{equation*} -$$ -

       
      - -Note that \( \hat{r}_k \) is the negative gradient of \( f \) at -\( \hat{x}=\hat{x}_k \), -so the gradient descent method would be to move in the direction \( \hat{r}_k \). -Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other, -so we take the direction closest to the gradient \( \hat{r}_k \) -under the conjugacy constraint. -This gives the following expression -

       
      -$$ -\begin{equation*} -\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k. -\end{equation*} -$$ -

       
      -

      -
      - - -
      -

      Conjugate gradient method

      -
      - -

      -We can also compute the residual iteratively as -

       
      -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, - \end{equation*} -$$ -

       
      - -which equals -

       
      -$$ -\begin{equation*} -\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k), - \end{equation*} -$$ -

       
      - -or -

       
      -$$ -\begin{equation*} -(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k, - \end{equation*} -$$ -

       
      - -which gives - -

       
      -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k}, - \end{equation*} -$$ -

       
      -

      -
      - - -
      -

      Simple implementation of the Conjugate gradient algorithm

      -
      - -

      - - -

        Vector ConjugateGradient(Matrix A, Vector b, Vector x0){
      -  int dim = x0.Dimension();
      -  const double tolerance = 1.0e-14;
      -  Vector x(dim),r(dim),v(dim),z(dim);
      -  double c,t,d;
      -
      -  x = x0;
      -  r = b - A*x;
      -  v = r;
      -  c = dot(r,r);
      -  int i = 0; IterMax = dim;
      -  while(i <= IterMax){
      -    z = A*v;
      -    t = c/dot(v,z);
      -    x = x + t*v;
      -    r = r - t*z;
      -    d = dot(r,r);
      -    if(sqrt(d) < tolerance)
      -      break;
      -    v = r + (d/c)*v;
      -    c = d;  i++;
      -  }
      -  return x;
      -} 
      -
      - -
      -
      - - -
      -

      Broyden–Fletcher–Goldfarb–Shanno algorithm

      -
      - -

      -The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take. - -

      -The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage. - -

      -The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation -

       
      -$$ -B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}), -$$ -

       
      - -

      -where \( B_{k} \) is an approximation to the Hessian matrix, which is -updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \) -is the gradient of the function -evaluated at \( x_k \). -A line search in the direction \( p_k \) is then used to -find the next point \( x_{k+1} \) by minimising -

       
      -$$ -f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}), -$$ -

       
      - -over the scalar \( \alpha > 0 \). - - -

      -
      - - -
      -

      Revisiting our first homework

      +

      Revisiting our first homework

      We will use linear regression as a case study for the gradient descent @@ -1321,7 +667,7 @@ $$

      -

      Gradient descent example

      +

      Gradient descent example

      Let \( \mathbf{y} = (y_1,\cdots,y_n)^T \), \( \mathbf{\hat{y}} = (\hat{y}_1,\cdots,\hat{y}_n)^T \) and \( \beta = (\beta_0, \beta_1)^T \) @@ -1350,7 +696,7 @@ and we want to find \( \beta \) such that \( C(\beta) \) is minimized.

      -

      The derivative of the cost/loss function

      +

      The derivative of the cost/loss function

      Computing \( \partial C(\beta) / \partial \beta_0 \) and \( \partial C(\beta) / \partial \beta_1 \) we can show that the gradient can be written as @@ -1367,7 +713,7 @@ where \( X \) is the design matrix defined above.

      -

      The Hessian matrix

      +

      The Hessian matrix

      The Hessian matrix of \( C(\beta) \) is given by

       
      $$ @@ -1383,7 +729,7 @@ This result implies that \( C(\beta) \) is a convex function since the matrix \(

      -

      Simple program

      +

      Simple program

      We can now write a program that minimizes \( C(\beta) \) using the gradient descent method with a constant learning rate \( \gamma \) according to @@ -1427,7 +773,7 @@ beta_NE = np.dot(Xt_X_inv,Xt_y)

      -

      Gradient Descent Example

      +

      Gradient Descent Example

      Another simple example is here @@ -1477,7 +823,7 @@ plt.show()

      -

      And a corresponding example using scikit-learn

      +

      And a corresponding example using scikit-learn

      @@ -1502,7 +848,7 @@ sgdreg.fit(x,y.ravel())

      -

      Gradient descent and Ridge

      +

      Gradient descent and Ridge

      We have also discussed Ridge regression where the loss function contains a regularized given by the \( L_2 \) norm of \( \beta \), @@ -1566,404 +912,7 @@ beta_ridge = np.dot(Z,np.dot(X.T,y))

      -

      Automatic differentiation

      -Python has tools for so-called automatic differentiation. -Consider the following example -

       
      -$$ -f(x) = \sin\left(2\pi x + x^2\right) -$$ -

       
      - -which has the following derivative -

       
      -$$ -f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) -$$ -

       
      - -Using autograd we have - -

      - - -

      import autograd.numpy as np
      -
      -# To do elementwise differentiation:
      -from autograd import elementwise_grad as egrad 
      -
      -# To plot:
      -import matplotlib.pyplot as plt 
      -
      -
      -def f(x):
      -    return np.sin(2*np.pi*x + x**2)
      -
      -def f_grad_analytic(x):
      -    return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
      -
      -# Do the comparison:
      -x = np.linspace(0,1,1000)
      -
      -f_grad = egrad(f)
      -
      -computed = f_grad(x)
      -analytic = f_grad_analytic(x)
      -
      -plt.title('Derivative computed from Autograd compared with the analytical derivative')
      -plt.plot(x,computed,label='autograd')
      -plt.plot(x,analytic,label='analytic')
      -
      -plt.xlabel('x')
      -plt.ylabel('y')
      -plt.legend()
      -
      -plt.show()
      -
      -print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
      -
      -
      - - -
      -

      Using autograd

      - -

      -Here we -experiment with what kind of functions Autograd is capable -of finding the gradient of. The following Python functions are just -meant to illustrate what Autograd can do, but please feel free to -experiment with other, possibly more complicated, functions as well. - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -
      -def f1(x):
      -    return x**3 + 1
      -
      -f1_grad = grad(f1)
      -
      -# Remember to send in float as argument to the computed gradient from Autograd!
      -a = 1.0
      -
      -# See the evaluated gradient at a using autograd:
      -print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
      -
      -# Compare with the analytical derivative, that is f1'(x) = 3*x**2 
      -grad_analytical = 3*a**2
      -print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
      -
      -
      - - -
      -

      Autograd with more complicated functions

      - -

      -To differentiate with respect to two (or more) arguments of a Python -function, Autograd need to know at which variable the function if -being differentiated with respect to. - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f2(x1,x2):
      -    return 3*x1**3 + x2*(x1 - 5) + 1
      -
      -# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
      -f2_grad_x1 = grad(f2,0)
      -
      -# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
      -f2_grad_x2 = grad(f2,1)
      -
      -x1 = 1.0
      -x2 = 3.0 
      -
      -print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
      -print("-"*30)
      -
      -# Compare with the analytical derivatives:
      -
      -# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
      -f2_grad_x1_analytical = 9*x1**2 + x2
      -
      -# Derivative of f2 w.r.t x2 is: x1 - 5:
      -f2_grad_x2_analytical = x1 - 5
      -
      -# See the evaluated derivations:
      -print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
      -print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
      -
      -print()
      -
      -print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
      -print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
      -
      -

      -Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable. -

      - - -
      -

      More complicated functions using the elements of their arguments directly

      - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f3(x): # Assumes x is an array of length 5 or higher
      -    return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
      -
      -f3_grad = grad(f3)
      -
      -x = np.linspace(0,4,5)
      -
      -# Print the computed gradient:
      -print("The computed gradient of f3 is: ", f3_grad(x))
      -
      -# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
      -f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
      -
      -# Print the analytical gradient:
      -print("The analytical gradient of f3 is: ", f3_grad_analytical)
      -
      -

      -Note that in this case, when sending an array as input argument, the -output from Autograd is another array. This is the true gradient of -the function, as opposed to the function in the previous example. By -using arrays to represent the variables, the output from Autograd -might be easier to work with, as the output is closer to what one -could expect form a gradient-evaluting function. -

      - - -
      -

      Functions using mathematical functions from Numpy

      - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f4(x):
      -    return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
      -
      -f4_grad = grad(f4)
      -
      -x = 2.7
      -
      -# Print the computed derivative:
      -print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
      -
      -# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
      -f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
      -
      -# Print the analytical gradient:
      -print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
      -
      -
      - - -
      -

      More autograd

      - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f5(x):
      -    if x >= 0:
      -        return x**2
      -    else:
      -        return -3*x + 1
      -
      -f5_grad = grad(f5)
      -
      -x = 2.7
      -
      -# Print the computed derivative:
      -print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
      -
      -
      - - -
      -

      And with loops

      - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f6_for(x):
      -    val = 0
      -    for i in range(10):
      -        val = val + x**i
      -    return val
      -
      -def f6_while(x):
      -    val = 0
      -    i = 0
      -    while i < 10:
      -        val = val + x**i
      -        i = i + 1
      -    return val
      -
      -f6_for_grad = grad(f6_for)
      -f6_while_grad = grad(f6_while)
      -
      -x = 0.5
      -
      -# Print the computed derivaties of f6_for and f6_while
      -print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
      -print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
      -
      -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
      -# The analytical derivative is: sum(i*x**(i-1)) 
      -f6_grad_analytical = 0
      -for i in range(10):
      -    f6_grad_analytical += i*x**(i-1)
      -
      -print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
      -
      -
      - - -
      -

      Using recursion

      -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -
      -def f7(n): # Assume that n is an integer
      -    if n == 1 or n == 0:
      -        return 1
      -    else:
      -        return n*f7(n-1)
      -
      -f7_grad = grad(f7)
      -
      -n = 2.0
      -
      -print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
      -
      -# The function f7 is an implementation of the factorial of n.
      -# By using the product rule, one can find that the derivative is:
      -
      -f7_grad_analytical = 0
      -for i in range(int(n)-1):
      -    tmp = 1
      -    for k in range(int(n)-1):
      -        if k != i:
      -            tmp *= (n - k)
      -    f7_grad_analytical += tmp
      -
      -print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
      -
      -

      -Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input. -

      - - -
      -

      Unsupported functions

      -Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd. - -

      -Assigning a value to the variable being differentiated with respect to -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f8(x): # Assume x is an array
      -    x[2] = 3
      -    return x*2
      -
      -f8_grad = grad(f8)
      -
      -x = 8.4
      -
      -print("The derivative of f8 is:",f8_grad(x))
      -
      -

      -Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible. -

      - - -
      -

      The syntax a.dot(b) when finding the dot product

      -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f9(a): # Assume a is an array with 2 elements
      -    b = np.array([1.0,2.0])
      -    return a.dot(b)
      -
      -f9_grad = grad(f9)
      -
      -x = np.array([1.0,0.0])
      -
      -print("The derivative of f9 is:",f9_grad(x))
      -
      -

      -Here we are told that the 'dot' function does not belong to Autograd's -version of a Numpy array. To overcome this, an alternative syntax -which also computed the dot product can be used: - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f9_alternative(x): # Assume a is an array with 2 elements
      -    b = np.array([1.0,2.0])
      -    return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
      -
      -f9_alternative_grad = grad(f9_alternative)
      -
      -x = np.array([3.0,0.0])
      -
      -print("The gradient of f9 is:",f9_alternative_grad(x))
      -
      -# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
      -# w.r.t x is (b_1, b_2).
      -
      -
      - - -
      -

      Recommended to avoid

      -The documentation recommends to avoid inplace operations such as -

      - - -

      a += b
      -a -= b
      -a*= b
      -a /=b
      -
      -
      - - -
      -

      Stochastic Gradient Descent

      +

      Stochastic Gradient Descent

      Stochastic gradient descent (SGD) and variants thereof address some of @@ -1983,7 +932,7 @@ $$

      -

      Computation of gradients

      +

      Computation of gradients

      This in turn means that the gradient can be @@ -2005,7 +954,7 @@ minibatches. We denote these minibatches by \( B_k \) where

      -

      SGD example

      +

      SGD example

      As an example, suppose we have \( 10 \) data points \( (\mathbf{x}_1,\cdots, \mathbf{x}_{10}) \) and we choose to have \( M=5 \) minibathces, then each minibatch contains two data points. In particular we have @@ -2031,7 +980,7 @@ $$
      -

      The gradient step

      +

      The gradient step

      Thus a gradient descent step now looks like @@ -2052,7 +1001,7 @@ the number of minibatches, as exemplified in the code below.

      -

      Simple example code

      +

      Simple example code

      @@ -2084,7 +1033,7 @@ all \( n \) datapoints.

      -

      When do we stop?

      +

      When do we stop?

      A natural question is when do we stop the search for a new minimum? @@ -2101,7 +1050,7 @@ gave the lowest value.

      -

      Slightly different approach

      +

      Slightly different approach

      Another approach is to let the step length \( \gamma_j \) depend on the @@ -2152,7 +1101,7 @@ j = 0

      -

      Program for stochastic gradient

      +

      Program for stochastic gradient

      @@ -2232,7 +1181,7 @@ plt.show()

      -

      Using gradient descent methods, limitations

      +

      Using gradient descent methods, limitations

      • Gradient descent (GD) finds local minima of our function. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our energy function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.
      • @@ -2246,7 +1195,7 @@ plt.show()
        -

        Momentum based GD

        +

        Momentum based GD

        The stochastic gradient descent (SGD) is almost always used with a momentum or inertia term that serves as a memory of the direction we are moving in parameter space. This is typically @@ -2273,7 +1222,7 @@ where we have defined \( \Delta \boldsymbol{\theta}_{t}= \boldsymbol{\theta}_t-\

        -

        More on momentum based approaches

        +

        More on momentum based approaches

        Let us try to get more intuition from these equations. It is helpful to consider a simple physical analogy with a particle of mass \( m \) moving in a viscous medium with drag coefficient \( \mu \) and potential @@ -2301,7 +1250,7 @@ $$

        -

        Momentum parameter

        +

        Momentum parameter

        Notice that this equation is identical to previous one if we identify the position of the particle, \( \mathbf{w} \), with the parameters \( \boldsymbol{\theta} \). This allows us to identify the momentum parameter and learning rate with the mass of the particle and the viscous drag as:

         
        @@ -2335,7 +1284,7 @@ One of the major advantages of NAG is that it allows for the use of a larger lea

        -

        Second moment of the gradient

        +

        Second moment of the gradient

        In stochastic gradient descent, with and without momentum, we still @@ -2360,7 +1309,7 @@ Recently, a number of methods have been introduced that accomplish this by track

        -

        RMS prop

        +

        RMS prop

        In RMS prop, in addition to keeping a running average of the first moment of the gradient, we also keep track of the second moment denoted by \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}_t^2] \). The update rule for RMS prop is given by @@ -2380,7 +1329,7 @@ where \( \beta \) controls the averaging time of the second moment and is typica

        -

        ADAM optimizer

        +

        ADAM optimizer

        A related algorithm is the ADAM optimizer. In ADAM, we keep a running average of both the first and second moment of the gradient and use this information to adaptively change the learning rate for different parameters. In addition to keeping a running average of the first and second moments of the gradient (i.e. \( \mathbf{m}_t=\mathbb{E}[\mathbf{g}_t] \) and \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}^2_t] \), respectively), ADAM performs an additional bias correction to account for the fact that we are estimating the first two moments of the gradient using a running average (denoted by the hats in the update rule below). The update rule for ADAM is given by (where multiplication and division are once again understood to be element-wise operations below) @@ -2412,7 +1361,7 @@ $$

        -

        Practical tips

        +

        Practical tips

        • Randomize the data when making mini-batches. It is always important to randomly shuffle the data when forming mini-batches. Otherwise, the gradient descent method can fit spurious correlations resulting from the order in which data is presented.
        • @@ -2426,6 +1375,1062 @@ Geron's text, see chapter 11, has several interesting discussions.
        +
        +

        Automatic differentiation

        +Python has tools for so-called automatic differentiation. +Consider the following example +

         
        +$$ +f(x) = \sin\left(2\pi x + x^2\right) +$$ +

         
        + +which has the following derivative +

         
        +$$ +f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) +$$ +

         
        + +Using autograd we have + +

        + + +

        import autograd.numpy as np
        +
        +# To do elementwise differentiation:
        +from autograd import elementwise_grad as egrad 
        +
        +# To plot:
        +import matplotlib.pyplot as plt 
        +
        +
        +def f(x):
        +    return np.sin(2*np.pi*x + x**2)
        +
        +def f_grad_analytic(x):
        +    return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
        +
        +# Do the comparison:
        +x = np.linspace(0,1,1000)
        +
        +f_grad = egrad(f)
        +
        +computed = f_grad(x)
        +analytic = f_grad_analytic(x)
        +
        +plt.title('Derivative computed from Autograd compared with the analytical derivative')
        +plt.plot(x,computed,label='autograd')
        +plt.plot(x,analytic,label='analytic')
        +
        +plt.xlabel('x')
        +plt.ylabel('y')
        +plt.legend()
        +
        +plt.show()
        +
        +print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
        +
        +
        + + +
        +

        Using autograd

        + +

        +Here we +experiment with what kind of functions Autograd is capable +of finding the gradient of. The following Python functions are just +meant to illustrate what Autograd can do, but please feel free to +experiment with other, possibly more complicated, functions as well. + +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +
        +def f1(x):
        +    return x**3 + 1
        +
        +f1_grad = grad(f1)
        +
        +# Remember to send in float as argument to the computed gradient from Autograd!
        +a = 1.0
        +
        +# See the evaluated gradient at a using autograd:
        +print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
        +
        +# Compare with the analytical derivative, that is f1'(x) = 3*x**2 
        +grad_analytical = 3*a**2
        +print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
        +
        +
        + + +
        +

        Autograd with more complicated functions

        + +

        +To differentiate with respect to two (or more) arguments of a Python +function, Autograd need to know at which variable the function if +being differentiated with respect to. + +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +def f2(x1,x2):
        +    return 3*x1**3 + x2*(x1 - 5) + 1
        +
        +# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
        +f2_grad_x1 = grad(f2,0)
        +
        +# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
        +f2_grad_x2 = grad(f2,1)
        +
        +x1 = 1.0
        +x2 = 3.0 
        +
        +print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
        +print("-"*30)
        +
        +# Compare with the analytical derivatives:
        +
        +# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
        +f2_grad_x1_analytical = 9*x1**2 + x2
        +
        +# Derivative of f2 w.r.t x2 is: x1 - 5:
        +f2_grad_x2_analytical = x1 - 5
        +
        +# See the evaluated derivations:
        +print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
        +print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
        +
        +print()
        +
        +print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
        +print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
        +
        +

        +Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable. +

        + + +
        +

        More complicated functions using the elements of their arguments directly

        + +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +def f3(x): # Assumes x is an array of length 5 or higher
        +    return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
        +
        +f3_grad = grad(f3)
        +
        +x = np.linspace(0,4,5)
        +
        +# Print the computed gradient:
        +print("The computed gradient of f3 is: ", f3_grad(x))
        +
        +# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
        +f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
        +
        +# Print the analytical gradient:
        +print("The analytical gradient of f3 is: ", f3_grad_analytical)
        +
        +

        +Note that in this case, when sending an array as input argument, the +output from Autograd is another array. This is the true gradient of +the function, as opposed to the function in the previous example. By +using arrays to represent the variables, the output from Autograd +might be easier to work with, as the output is closer to what one +could expect form a gradient-evaluting function. +

        + + +
        +

        Functions using mathematical functions from Numpy

        + +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +def f4(x):
        +    return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
        +
        +f4_grad = grad(f4)
        +
        +x = 2.7
        +
        +# Print the computed derivative:
        +print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
        +
        +# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
        +f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
        +
        +# Print the analytical gradient:
        +print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
        +
        +
        + + +
        +

        More autograd

        + +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +def f5(x):
        +    if x >= 0:
        +        return x**2
        +    else:
        +        return -3*x + 1
        +
        +f5_grad = grad(f5)
        +
        +x = 2.7
        +
        +# Print the computed derivative:
        +print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
        +
        +
        + + +
        +

        And with loops

        + +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +def f6_for(x):
        +    val = 0
        +    for i in range(10):
        +        val = val + x**i
        +    return val
        +
        +def f6_while(x):
        +    val = 0
        +    i = 0
        +    while i < 10:
        +        val = val + x**i
        +        i = i + 1
        +    return val
        +
        +f6_for_grad = grad(f6_for)
        +f6_while_grad = grad(f6_while)
        +
        +x = 0.5
        +
        +# Print the computed derivaties of f6_for and f6_while
        +print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
        +print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
        +
        +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
        +# The analytical derivative is: sum(i*x**(i-1)) 
        +f6_grad_analytical = 0
        +for i in range(10):
        +    f6_grad_analytical += i*x**(i-1)
        +
        +print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
        +
        +
        + + +
        +

        Using recursion

        +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +
        +def f7(n): # Assume that n is an integer
        +    if n == 1 or n == 0:
        +        return 1
        +    else:
        +        return n*f7(n-1)
        +
        +f7_grad = grad(f7)
        +
        +n = 2.0
        +
        +print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
        +
        +# The function f7 is an implementation of the factorial of n.
        +# By using the product rule, one can find that the derivative is:
        +
        +f7_grad_analytical = 0
        +for i in range(int(n)-1):
        +    tmp = 1
        +    for k in range(int(n)-1):
        +        if k != i:
        +            tmp *= (n - k)
        +    f7_grad_analytical += tmp
        +
        +print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
        +
        +

        +Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input. +

        + + +
        +

        Unsupported functions

        +Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd. + +

        +Assigning a value to the variable being differentiated with respect to +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +def f8(x): # Assume x is an array
        +    x[2] = 3
        +    return x*2
        +
        +f8_grad = grad(f8)
        +
        +x = 8.4
        +
        +print("The derivative of f8 is:",f8_grad(x))
        +
        +

        +Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible. +

        + + +
        +

        The syntax a.dot(b) when finding the dot product

        +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +def f9(a): # Assume a is an array with 2 elements
        +    b = np.array([1.0,2.0])
        +    return a.dot(b)
        +
        +f9_grad = grad(f9)
        +
        +x = np.array([1.0,0.0])
        +
        +print("The derivative of f9 is:",f9_grad(x))
        +
        +

        +Here we are told that the 'dot' function does not belong to Autograd's +version of a Numpy array. To overcome this, an alternative syntax +which also computed the dot product can be used: + +

        + + +

        import autograd.numpy as np
        +from autograd import grad
        +def f9_alternative(x): # Assume a is an array with 2 elements
        +    b = np.array([1.0,2.0])
        +    return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
        +
        +f9_alternative_grad = grad(f9_alternative)
        +
        +x = np.array([3.0,0.0])
        +
        +print("The gradient of f9 is:",f9_alternative_grad(x))
        +
        +# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
        +# w.r.t x is (b_1, b_2).
        +
        +
        + + +
        +

        Recommended to avoid

        +The documentation recommends to avoid inplace operations such as +

        + + +

        a += b
        +a -= b
        +a*= b
        +a /=b
        +
        +
        + + +
        +

        Standard steepest descent

        + +

        +Before we proceed, we would like to discuss the approach called the +standard Steepest descent, which again leads to us having to be able +to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG). + +

        +The success of the CG method +for finding solutions of non-linear problems is based on the theory +of conjugate gradients for linear systems of equations. It belongs to +the class of iterative methods for solving problems from linear +algebra of the type +

         
        +$$ +\begin{equation*} +\hat{A}\hat{x} = \hat{b}. +\end{equation*} +$$ +

         
        + +

        +In the iterative process we end up with a problem like + +

         
        +$$ +\begin{equation*} + \hat{r}= \hat{b}-\hat{A}\hat{x}, +\end{equation*} +$$ +

         
        + +where \( \hat{r} \) is the so-called residual or error in the iterative process. + +

        +When we have found the exact solution, \( \hat{r}=0 \). +

        + + +
        +

        Gradient method

        + +

        +The residual is zero when we reach the minimum of the quadratic equation +

         
        +$$ +\begin{equation*} + P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b}, +\end{equation*} +$$ +

         
        + +

        +with the constraint that the matrix \( \hat{A} \) is positive definite and +symmetric. This defines also the Hessian and we want it to be positive definite. +

        + + +
        +

        Steepest descent method

        + +

        +We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). +We can assume without loss of generality that +

         
        +$$ +\begin{equation*} +\hat{x}_0=0, +\end{equation*} +$$ +

         
        + +or consider the system +

         
        +$$ +\begin{equation*} +\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, +\end{equation*} +$$ +

         
        + +instead. +

        + + +
        +

        Steepest descent method

        +
        + +

        +One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form +

         
        +$$ +\begin{equation*} + f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. +\end{equation*} +$$ +

         
        + +This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition) +to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), +which equals +

         
        +$$ +\begin{equation*} +\hat{A}\hat{x}_0-\hat{b}, +\end{equation*} +$$ +

         
        + +and +\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). + + +

        +
        + + +
        +

        Final expressions

        +
        + +

        +We can compute the residual iteratively as +

         
        +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, + \end{equation*} +$$ +

         
        + +which equals +

         
        +$$ +\begin{equation*} +\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k), + \end{equation*} +$$ +

         
        + +or +

         
        +$$ +\begin{equation*} +(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k, + \end{equation*} +$$ +

         
        + +which gives + +

         
        +$$ +\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k} +$$ +

         
        + +leading to the iterative scheme +

         
        +$$ +\begin{equation*} +\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k}, + \end{equation*} +$$ +

         
        +

        +
        + + +
        +

        Code examples for steepest descent

        +
        + + +
        +

        Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come

        +
        + +

        + + +

        #include <cmath>
        +#include <iostream>
        +#include <fstream>
        +#include <iomanip>
        +#include "vectormatrixclass.h"
        +using namespace  std;
        +//   Main function begins here
        +int main(int  argc, char * argv[]){
        +  int dim = 2;
        +  Vector x(dim),xsd(dim), b(dim),x0(dim);
        +  Matrix A(dim,dim);
        +
        +  // Set our initial guess
        +  x0(0) = x0(1) = 0;
        +  // Set the matrix
        +  A(0,0) =  3;    A(1,0) =  2;   A(0,1) =  2;   A(1,1) =  6;
        +  b(0) = 2; b(1) = -8;
        +  cout << "The Matrix A that we are using: " << endl;
        +  A.Print();
        +  cout << endl;
        +  xsd = SteepestDescent(A,b,x0);
        +  cout << "The approximate solution using Steepest Descent is: " << endl;
        +  xsd.Print();
        +  cout << endl;
        +}
        +
        + +
        +
        + + +
        +

        The routine for the steepest descent method

        +
        + +

        + + +

        Vector SteepestDescent(Matrix A, Vector b, Vector x0){
        +  int IterMax, i;
        +  int dim = x0.Dimension();
        +  const double tolerance = 1.0e-14;
        +  Vector x(dim),f(dim),z(dim);
        +  double c,alpha,d;
        +  IterMax = 30;
        +  x = x0;
        +  r = A*x-b;
        +  i = 0;
        +  while (i <= IterMax){
        +    z = A*r;
        +    c = dot(r,r);
        +    alpha = c/dot(r,z);
        +    x = x - alpha*r;
        +    r =  A*x-b;
        +    if(sqrt(dot(r,r)) < tolerance) break;
        +    i++;
        +  }
        +  return x;
        +}
        +
        + +
        +
        + + +
        +

        Steepest descent example

        + +

        + + +

        import numpy as np
        +import numpy.linalg as la
        +
        +import scipy.optimize as sopt
        +
        +import matplotlib.pyplot as pt
        +from mpl_toolkits.mplot3d import axes3d
        +
        +def f(x):
        +    return 0.5*x[0]**2 + 2.5*x[1]**2
        +
        +def df(x):
        +    return np.array([x[0], 5*x[1]])
        +
        +fig = pt.figure()
        +ax = fig.gca(projection="3d")
        +
        +xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]
        +fmesh = f(np.array([xmesh, ymesh]))
        +ax.plot_surface(xmesh, ymesh, fmesh)
        +
        +

        +And then as countor plot +

        + + +

        pt.axis("equal")
        +pt.contour(xmesh, ymesh, fmesh)
        +guesses = [np.array([2, 2./5])]
        +
        +

        +Find guesses +

        + + +

        x = guesses[-1]
        +s = -df(x)
        +
        +

        +Run it! +

        + + +

        def f1d(alpha):
        +    return f(x + alpha*s)
        +
        +alpha_opt = sopt.golden(f1d)
        +next_guess = x + alpha_opt * s
        +guesses.append(next_guess)
        +print(next_guess)
        +
        +

        +What happened? +

        + + +

        pt.axis("equal")
        +pt.contour(xmesh, ymesh, fmesh, 50)
        +it_array = np.array(guesses)
        +pt.plot(it_array.T[0], it_array.T[1], "x-")
        +
        +
        + + +
        +

        Conjugate gradient method

        +
        + +

        +In the CG method we define so-called conjugate directions and two vectors +\( \hat{s} \) and \( \hat{t} \) +are said to be +conjugate if +

         
        +$$ +\begin{equation*} +\hat{s}^T\hat{A}\hat{t}= 0. +\end{equation*} +$$ +

         
        + +The philosophy of the CG method is to perform searches in various conjugate directions +of our vectors \( \hat{x}_i \) obeying the above criterion, namely +

         
        +$$ +\begin{equation*} +\hat{x}_i^T\hat{A}\hat{x}_j= 0. +\end{equation*} +$$ +

         
        + +Two vectors are conjugate if they are orthogonal with respect to +this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \). +

        +
        + + +
        +

        Conjugate gradient method

        +
        + +

        +An example is given by the eigenvectors of the matrix +

         
        +$$ +\begin{equation*} +\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j, +\end{equation*} +$$ +

         
        + +which is zero unless \( i=j \). +

        +
        + + +
        +

        Conjugate gradient method

        +
        + +

        +Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size +\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector +

         
        +$$ +\begin{equation*} +\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}. +\end{equation*} +$$ +

         
        + +We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions. +Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution +$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely + +

         
        +$$ +\begin{equation*} + \hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i. +\end{equation*} +$$ +

         
        +

        +
        + + +
        +

        Conjugate gradient method

        +
        + +

        +The coefficients are given by +

         
        +$$ +\begin{equation*} + \mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. +\end{equation*} +$$ +

         
        + +Multiplying with \( \hat{p}_k^T \) from the left gives + +

         
        +$$ +\begin{equation*} + \hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b}, +\end{equation*} +$$ +

         
        + +and we can define the coefficients \( \alpha_k \) as + +

         
        +$$ +\begin{equation*} + \alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k} +\end{equation*} +$$ +

         
        +

        +
        + + +
        +

        Conjugate gradient method and iterations

        +
        + +

        +If we choose the conjugate vectors \( \hat{p}_k \) carefully, +then we may not need all of them to obtain a good approximation to the solution +\( \hat{x} \). +We want to regard the conjugate gradient method as an iterative method. +This will us to solve systems where \( n \) is so large that the direct +method would take too much time. + +

        +We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). +We can assume without loss of generality that +

         
        +$$ +\begin{equation*} +\hat{x}_0=0, +\end{equation*} +$$ +

         
        + +or consider the system +

         
        +$$ +\begin{equation*} +\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, +\end{equation*} +$$ +

         
        + +instead. +

        +
        + + +
        +

        Conjugate gradient method

        +
        + +

        +One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form +

         
        +$$ +\begin{equation*} + f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. +\end{equation*} +$$ +

         
        + +This suggests taking the first basis vector \( \hat{p}_1 \) +to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), +which equals +

         
        +$$ +\begin{equation*} +\hat{A}\hat{x}_0-\hat{b}, +\end{equation*} +$$ +

         
        + +and +\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). +The other vectors in the basis will be conjugate to the gradient, +hence the name conjugate gradient method. +

        +
        + + +
        +

        Conjugate gradient method

        +
        + +

        +Let \( \hat{r}_k \) be the residual at the \( k \)-th step: +

         
        +$$ +\begin{equation*} +\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k. +\end{equation*} +$$ +

         
        + +Note that \( \hat{r}_k \) is the negative gradient of \( f \) at +\( \hat{x}=\hat{x}_k \), +so the gradient descent method would be to move in the direction \( \hat{r}_k \). +Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other, +so we take the direction closest to the gradient \( \hat{r}_k \) +under the conjugacy constraint. +This gives the following expression +

         
        +$$ +\begin{equation*} +\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k. +\end{equation*} +$$ +

         
        +

        +
        + + +
        +

        Conjugate gradient method

        +
        + +

        +We can also compute the residual iteratively as +

         
        +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, + \end{equation*} +$$ +

         
        + +which equals +

         
        +$$ +\begin{equation*} +\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k), + \end{equation*} +$$ +

         
        + +or +

         
        +$$ +\begin{equation*} +(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k, + \end{equation*} +$$ +

         
        + +which gives + +

         
        +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k}, + \end{equation*} +$$ +

         
        +

        +
        + + +
        +

        Simple implementation of the Conjugate gradient algorithm

        +
        + +

        + + +

          Vector ConjugateGradient(Matrix A, Vector b, Vector x0){
        +  int dim = x0.Dimension();
        +  const double tolerance = 1.0e-14;
        +  Vector x(dim),r(dim),v(dim),z(dim);
        +  double c,t,d;
        +
        +  x = x0;
        +  r = b - A*x;
        +  v = r;
        +  c = dot(r,r);
        +  int i = 0; IterMax = dim;
        +  while(i <= IterMax){
        +    z = A*v;
        +    t = c/dot(v,z);
        +    x = x + t*v;
        +    r = r - t*z;
        +    d = dot(r,r);
        +    if(sqrt(d) < tolerance)
        +      break;
        +    v = r + (d/c)*v;
        +    c = d;  i++;
        +  }
        +  return x;
        +} 
        +
        + +
        +
        + + +
        +

        Broyden–Fletcher–Goldfarb–Shanno algorithm

        +
        + +

        +The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take. + +

        +The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage. + +

        +The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation +

         
        +$$ +B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}), +$$ +

         
        + +

        +where \( B_{k} \) is an approximation to the Hessian matrix, which is +updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \) +is the gradient of the function +evaluated at \( x_k \). +A line search in the direction \( p_k \) is then used to +find the next point \( x_{k+1} \) by minimising +

         
        +$$ +f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}), +$$ +

         
        + +over the scalar \( \alpha > 0 \). + + +

        +
        + +
    diff --git a/doc/pub/Splines/html/Splines-solarized.html b/doc/pub/Splines/html/Splines-solarized.html index 380b5b670..ca6b66717 100644 --- a/doc/pub/Splines/html/Splines-solarized.html +++ b/doc/pub/Splines/html/Splines-solarized.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -60,113 +61,114 @@ div { text-align: justify; text-justify: inter-word; } @@ -208,12 +210,17 @@ MathJax.Hub.Config({
    [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

    -

    Oct 18, 2018

    +

    Sep 19, 2019












    -

    Optimization, the central part of any Machine Learning algortithm

    +

    Optimization problems, why?

    + +

    +









    + +

    Optimization, the central part of any Machine Learning algortithm

    Almost every problem in machine learning and data science starts with @@ -228,7 +235,7 @@ some approximative/numerical method to compute the minimum.











    -

    Revisiting our Logistic Regression case

    +

    Revisiting our Logistic Regression case

    In our discussion on Logistic Regression we studied the @@ -250,7 +257,7 @@ where \( \hat{\beta} \) are the weights we wish to extract from data, in our cas











    -

    The equations to solve

    +

    The equations to solve

    Our compact equations used a definition of a vector \( \hat{y} \) with \( n \) @@ -276,7 +283,7 @@ This defines what is called the Hessian matrix.











    -

    Solving using Newton-Raphson's method

    +

    Solving using Newton-Raphson's method

    If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives. @@ -302,7 +309,7 @@ If we can compute these matrices, in particular the Hessian, the above is often











    -

    Brief reminder on Newton-Raphson's method

    +

    Brief reminder on Newton-Raphson's method

    Let us quickly remind ourselves how we derive the above method. @@ -319,7 +326,7 @@ normally discourage the use of this method.











    -

    The equations

    +

    The equations

    The Newton-Raphson formula consists geometrically of extending the @@ -355,7 +362,7 @@ $$











    -

    Simple geometric interpretation

    +

    Simple geometric interpretation

    The above is Newton-Raphson's method. It has a simple geometric @@ -373,7 +380,7 @@ vanishes, then Newton-Raphson may fail totally











    -

    Extending to more than one variable

    +

    Extending to more than one variable

    Newton's method can be generalized to systems of several non-linear equations @@ -428,7 +435,7 @@ more than two non-linear equations. In our case, the Jacobian matrix is given by











    -

    Steepest descent

    +

    Steepest descent

    The basic idea of gradient descent is @@ -452,7 +459,7 @@ we are always moving towards smaller function values, i.e a minimum.

    -

    More on Steepest descent

    +

    More on Steepest descent

    The previous observation is the basis of the method of steepest @@ -471,7 +478,7 @@ the learning rate within the context of Machine Learning.

    -

    The ideal

    +

    The ideal

    Ideally the sequence \( \{\mathbf{x}_k \}_{k=0} \) converges to a global @@ -497,7 +504,7 @@ Note that the gradient is a function of \( \mathbf{x} =

    -

    The sensitiveness of the gradient descent

    +

    The sensitiveness of the gradient descent

    The gradient descent method @@ -516,7 +523,7 @@ randomness. One such method is that of Stochastic Gradient Descent

    -

    Convex functions

    +

    Convex functions

    Ideally we want our cost/loss function to be convex(concave). @@ -536,7 +543,7 @@ regular polygons (triangles, rectangles, pentagons, etc...).











    -

    Convex function

    +

    Convex function

    Convex function: Let \( X \subset \mathbb{R}^n \) be a convex set. Assume that the function \( f: X \rightarrow \mathbb{R} \) is continuous, then \( f \) is said to be convex if $$f(tx_1 + (1-t)x_2) \leq tf(x_1) + (1-t)f(x_2) $$ for all \( x_1, x_2 \in X \) and for all \( t \in [0,1] \). If \( \leq \) is replaced with a strict inequaltiy in the definition, we demand \( x_1 \neq x_2 \) and \( t\in(0,1) \) then \( f \) is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting \( f(x_1) \) and \( f(x_2) \), the value of the function on the interval \( [x_1,x_2] \) is always below the line as illustrated below. @@ -544,7 +551,7 @@ regular polygons (triangles, rectangles, pentagons, etc...).











    -

    Conditions on convex functions

    +

    Conditions on convex functions

    In the following we state first and second-order conditions which @@ -586,7 +593,7 @@ This condition is particularly useful since it gives us an procedure for determi











    -

    More on convex functions

    +

    More on convex functions

    The next result is of great importance to us and the reason why we are @@ -616,7 +623,7 @@ This result means that if we know that the cost/loss function is convex and we a











    -

    Some simple problems

    +

    Some simple problems

    1. Show that \( f(x)=x^2 \) is convex for \( x \in \mathbb{R} \) using the definition of convexity. Hint: If you re-write the definition, \( f \) is convex if the following holds for all \( x,y \in D_f \) and any \( \lambda \in [0,1] \) $\lambda f(x)+(1-\lambda)f(y)-f(\lambda x + (1-\lambda) y ) \geq 0$.
    2. @@ -640,623 +647,10 @@ This result means that if we know that the cost/loss function is convex and we a Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this). -

      -









      - -

      Standard steepest descent

      - -

      -Before we proceed, we would like to discuss the approach called the -standard Steepest descent, which again leads to us having to be able -to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG). - -

      -The success of the CG method -for finding solutions of non-linear problems is based on the theory -of conjugate gradients for linear systems of equations. It belongs to -the class of iterative methods for solving problems from linear -algebra of the type -$$ -\begin{equation*} -\hat{A}\hat{x} = \hat{b}. -\end{equation*} -$$ - -

      -In the iterative process we end up with a problem like - -$$ -\begin{equation*} - \hat{r}= \hat{b}-\hat{A}\hat{x}, -\end{equation*} -$$ - -where \( \hat{r} \) is the so-called residual or error in the iterative process. - -

      -When we have found the exact solution, \( \hat{r}=0 \). - -

      -









      - -

      Gradient method

      - -

      -The residual is zero when we reach the minimum of the quadratic equation -$$ -\begin{equation*} - P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b}, -\end{equation*} -$$ - -

      -with the constraint that the matrix \( \hat{A} \) is positive definite and -symmetric. This defines also the Hessian and we want it to be positive definite. - -

      -









      - -

      Steepest descent method

      - -

      -We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). -We can assume without loss of generality that -$$ -\begin{equation*} -\hat{x}_0=0, -\end{equation*} -$$ - -or consider the system -$$ -\begin{equation*} -\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, -\end{equation*} -$$ - -instead. - -

      -









      - -

      Steepest descent method

      -
      - -

      -One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form -$$ -\begin{equation*} - f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. -\end{equation*} -$$ - -This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition) -to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), -which equals -$$ -\begin{equation*} -\hat{A}\hat{x}_0-\hat{b}, -\end{equation*} -$$ - -and -\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). - - -

      - - -

      -









      - -

      Final expressions

      -
      - -

      -We can compute the residual iteratively as -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, - \end{equation*} -$$ - -which equals -$$ -\begin{equation*} -\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k), - \end{equation*} -$$ - -or -$$ -\begin{equation*} -(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k, - \end{equation*} -$$ - -which gives - -$$ -\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k} -$$ - -leading to the iterative scheme -$$ -\begin{equation*} -\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k}, - \end{equation*} -$$ -

      - - -

      -









      - -

      Code examples for steepest descent

      - -

      -









      - -

      Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come

      -
      - -

      -

      - - -

      #include <cmath>
      -#include <iostream>
      -#include <fstream>
      -#include <iomanip>
      -#include "vectormatrixclass.h"
      -using namespace  std;
      -//   Main function begins here
      -int main(int  argc, char * argv[]){
      -  int dim = 2;
      -  Vector x(dim),xsd(dim), b(dim),x0(dim);
      -  Matrix A(dim,dim);
      -
      -  // Set our initial guess
      -  x0(0) = x0(1) = 0;
      -  // Set the matrix
      -  A(0,0) =  3;    A(1,0) =  2;   A(0,1) =  2;   A(1,1) =  6;
      -  b(0) = 2; b(1) = -8;
      -  cout << "The Matrix A that we are using: " << endl;
      -  A.Print();
      -  cout << endl;
      -  xsd = SteepestDescent(A,b,x0);
      -  cout << "The approximate solution using Steepest Descent is: " << endl;
      -  xsd.Print();
      -  cout << endl;
      -}
      -
      - -
      - - -

      -









      - -

      The routine for the steepest descent method

      -
      - -

      -

      - - -

      Vector SteepestDescent(Matrix A, Vector b, Vector x0){
      -  int IterMax, i;
      -  int dim = x0.Dimension();
      -  const double tolerance = 1.0e-14;
      -  Vector x(dim),f(dim),z(dim);
      -  double c,alpha,d;
      -  IterMax = 30;
      -  x = x0;
      -  r = A*x-b;
      -  i = 0;
      -  while (i <= IterMax){
      -    z = A*r;
      -    c = dot(r,r);
      -    alpha = c/dot(r,z);
      -    x = x - alpha*r;
      -    r =  A*x-b;
      -    if(sqrt(dot(r,r)) < tolerance) break;
      -    i++;
      -  }
      -  return x;
      -}
      -
      - -
      - - -

      -









      - -

      Steepest descent example

      - -

      - - -

      import numpy as np
      -import numpy.linalg as la
      -
      -import scipy.optimize as sopt
      -
      -import matplotlib.pyplot as pt
      -from mpl_toolkits.mplot3d import axes3d
      -
      -def f(x):
      -    return 0.5*x[0]**2 + 2.5*x[1]**2
      -
      -def df(x):
      -    return np.array([x[0], 5*x[1]])
      -
      -fig = pt.figure()
      -ax = fig.gca(projection="3d")
      -
      -xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]
      -fmesh = f(np.array([xmesh, ymesh]))
      -ax.plot_surface(xmesh, ymesh, fmesh)
      -
      -

      -And then as countor plot -

      - - -

      pt.axis("equal")
      -pt.contour(xmesh, ymesh, fmesh)
      -guesses = [np.array([2, 2./5])]
      -
      -

      -Find guesses -

      - - -

      x = guesses[-1]
      -s = -df(x)
      -
      -

      -Run it! -

      - - -

      def f1d(alpha):
      -    return f(x + alpha*s)
      -
      -alpha_opt = sopt.golden(f1d)
      -next_guess = x + alpha_opt * s
      -guesses.append(next_guess)
      -print(next_guess)
      -
      -

      -What happened? -

      - - -

      pt.axis("equal")
      -pt.contour(xmesh, ymesh, fmesh, 50)
      -it_array = np.array(guesses)
      -pt.plot(it_array.T[0], it_array.T[1], "x-")
      -
      -

      -









      - -

      Conjugate gradient method

      -
      - -

      -In the CG method we define so-called conjugate directions and two vectors -\( \hat{s} \) and \( \hat{t} \) -are said to be -conjugate if -$$ -\begin{equation*} -\hat{s}^T\hat{A}\hat{t}= 0. -\end{equation*} -$$ - -The philosophy of the CG method is to perform searches in various conjugate directions -of our vectors \( \hat{x}_i \) obeying the above criterion, namely -$$ -\begin{equation*} -\hat{x}_i^T\hat{A}\hat{x}_j= 0. -\end{equation*} -$$ - -Two vectors are conjugate if they are orthogonal with respect to -this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \). -

      - - -

      -









      - -

      Conjugate gradient method

      -
      - -

      -An example is given by the eigenvectors of the matrix -$$ -\begin{equation*} -\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j, -\end{equation*} -$$ - -which is zero unless \( i=j \). -

      - - -

      -









      - -

      Conjugate gradient method

      -
      - -

      -Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size -\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector -$$ -\begin{equation*} -\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}. -\end{equation*} -$$ - -We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions. -Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution -$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely - -$$ -\begin{equation*} - \hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i. -\end{equation*} -$$ -

      - - -

      -









      - -

      Conjugate gradient method

      -
      - -

      -The coefficients are given by -$$ -\begin{equation*} - \mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. -\end{equation*} -$$ - -Multiplying with \( \hat{p}_k^T \) from the left gives - -$$ -\begin{equation*} - \hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b}, -\end{equation*} -$$ - -and we can define the coefficients \( \alpha_k \) as - -$$ -\begin{equation*} - \alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k} -\end{equation*} -$$ -

      - - -

      -









      - -

      Conjugate gradient method and iterations

      -
      - -

      - -

      -If we choose the conjugate vectors \( \hat{p}_k \) carefully, -then we may not need all of them to obtain a good approximation to the solution -\( \hat{x} \). -We want to regard the conjugate gradient method as an iterative method. -This will us to solve systems where \( n \) is so large that the direct -method would take too much time. - -

      -We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). -We can assume without loss of generality that -$$ -\begin{equation*} -\hat{x}_0=0, -\end{equation*} -$$ - -or consider the system -$$ -\begin{equation*} -\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, -\end{equation*} -$$ - -instead. -

      - - -

      -









      - -

      Conjugate gradient method

      -
      - -

      -One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form -$$ -\begin{equation*} - f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. -\end{equation*} -$$ - -This suggests taking the first basis vector \( \hat{p}_1 \) -to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), -which equals -$$ -\begin{equation*} -\hat{A}\hat{x}_0-\hat{b}, -\end{equation*} -$$ - -and -\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). -The other vectors in the basis will be conjugate to the gradient, -hence the name conjugate gradient method. -

      - - -

      -









      - -

      Conjugate gradient method

      -
      - -

      -Let \( \hat{r}_k \) be the residual at the \( k \)-th step: -$$ -\begin{equation*} -\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k. -\end{equation*} -$$ - -Note that \( \hat{r}_k \) is the negative gradient of \( f \) at -\( \hat{x}=\hat{x}_k \), -so the gradient descent method would be to move in the direction \( \hat{r}_k \). -Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other, -so we take the direction closest to the gradient \( \hat{r}_k \) -under the conjugacy constraint. -This gives the following expression -$$ -\begin{equation*} -\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k. -\end{equation*} -$$ -

      - - -

      -









      - -

      Conjugate gradient method

      -
      - -

      -We can also compute the residual iteratively as -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, - \end{equation*} -$$ - -which equals -$$ -\begin{equation*} -\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k), - \end{equation*} -$$ - -or -$$ -\begin{equation*} -(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k, - \end{equation*} -$$ - -which gives - -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k}, - \end{equation*} -$$ -

      - - -

      -









      - -

      Simple implementation of the Conjugate gradient algorithm

      -
      - -

      -

      - - -

        Vector ConjugateGradient(Matrix A, Vector b, Vector x0){
      -  int dim = x0.Dimension();
      -  const double tolerance = 1.0e-14;
      -  Vector x(dim),r(dim),v(dim),z(dim);
      -  double c,t,d;
      -
      -  x = x0;
      -  r = b - A*x;
      -  v = r;
      -  c = dot(r,r);
      -  int i = 0; IterMax = dim;
      -  while(i <= IterMax){
      -    z = A*v;
      -    t = c/dot(v,z);
      -    x = x + t*v;
      -    r = r - t*z;
      -    d = dot(r,r);
      -    if(sqrt(d) < tolerance)
      -      break;
      -    v = r + (d/c)*v;
      -    c = d;  i++;
      -  }
      -  return x;
      -} 
      -
      - -
      - - -

      -









      - -

      Broyden–Fletcher–Goldfarb–Shanno algorithm

      -
      - -

      -The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take. - -

      -The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage. - -

      -The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation -$$ -B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}), -$$ - -

      -where \( B_{k} \) is an approximation to the Hessian matrix, which is -updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \) -is the gradient of the function -evaluated at \( x_k \). -A line search in the direction \( p_k \) is then used to -find the next point \( x_{k+1} \) by minimising -$$ -f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}), -$$ - -over the scalar \( \alpha > 0 \). - - -

      - -

      -

      Revisiting our first homework

      +

      Revisiting our first homework

      We will use linear regression as a case study for the gradient descent @@ -1289,7 +683,7 @@ $$

      -

      Gradient descent example

      +

      Gradient descent example

      Let \( \mathbf{y} = (y_1,\cdots,y_n)^T \), \( \mathbf{\hat{y}} = (\hat{y}_1,\cdots,\hat{y}_n)^T \) and \( \beta = (\beta_0, \beta_1)^T \) @@ -1314,7 +708,7 @@ and we want to find \( \beta \) such that \( C(\beta) \) is minimized.











      -

      The derivative of the cost/loss function

      +

      The derivative of the cost/loss function

      Computing \( \partial C(\beta) / \partial \beta_0 \) and \( \partial C(\beta) / \partial \beta_1 \) we can show that the gradient can be written as @@ -1329,7 +723,7 @@ where \( X \) is the design matrix defined above.











      -

      The Hessian matrix

      +

      The Hessian matrix

      The Hessian matrix of \( C(\beta) \) is given by $$ \hat{H} \equiv \begin{bmatrix} @@ -1343,7 +737,7 @@ This result implies that \( C(\beta) \) is a convex function since the matrix \(











      -

      Simple program

      +

      Simple program

      We can now write a program that minimizes \( C(\beta) \) using the gradient descent method with a constant learning rate \( \gamma \) according to @@ -1384,7 +778,7 @@ beta_NE = np.dot(Xt_X_inv,Xt_y)











      -

      Gradient Descent Example

      +

      Gradient Descent Example

      Another simple example is here @@ -1433,7 +827,7 @@ plt.show()











      -

      And a corresponding example using scikit-learn

      +

      And a corresponding example using scikit-learn

      @@ -1457,7 +851,7 @@ sgdreg.fit(x,y.ravel())

      -

      Gradient descent and Ridge

      +

      Gradient descent and Ridge

      We have also discussed Ridge regression where the loss function contains a regularized given by the \( L_2 \) norm of \( \beta \), @@ -1514,393 +908,7 @@ beta_ridge = np.dot(Z,np.dot(X.T,y))











      -

      Automatic differentiation

      -Python has tools for so-called automatic differentiation. -Consider the following example -$$ -f(x) = \sin\left(2\pi x + x^2\right) -$$ - -which has the following derivative -$$ -f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) -$$ - -Using autograd we have - -

      - - -

      import autograd.numpy as np
      -
      -# To do elementwise differentiation:
      -from autograd import elementwise_grad as egrad 
      -
      -# To plot:
      -import matplotlib.pyplot as plt 
      -
      -
      -def f(x):
      -    return np.sin(2*np.pi*x + x**2)
      -
      -def f_grad_analytic(x):
      -    return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
      -
      -# Do the comparison:
      -x = np.linspace(0,1,1000)
      -
      -f_grad = egrad(f)
      -
      -computed = f_grad(x)
      -analytic = f_grad_analytic(x)
      -
      -plt.title('Derivative computed from Autograd compared with the analytical derivative')
      -plt.plot(x,computed,label='autograd')
      -plt.plot(x,analytic,label='analytic')
      -
      -plt.xlabel('x')
      -plt.ylabel('y')
      -plt.legend()
      -
      -plt.show()
      -
      -print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
      -
      -

      - - -

      Using autograd

      - -

      -Here we -experiment with what kind of functions Autograd is capable -of finding the gradient of. The following Python functions are just -meant to illustrate what Autograd can do, but please feel free to -experiment with other, possibly more complicated, functions as well. - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -
      -def f1(x):
      -    return x**3 + 1
      -
      -f1_grad = grad(f1)
      -
      -# Remember to send in float as argument to the computed gradient from Autograd!
      -a = 1.0
      -
      -# See the evaluated gradient at a using autograd:
      -print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
      -
      -# Compare with the analytical derivative, that is f1'(x) = 3*x**2 
      -grad_analytical = 3*a**2
      -print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
      -
      -

      -









      - -

      Autograd with more complicated functions

      - -

      -To differentiate with respect to two (or more) arguments of a Python -function, Autograd need to know at which variable the function if -being differentiated with respect to. - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f2(x1,x2):
      -    return 3*x1**3 + x2*(x1 - 5) + 1
      -
      -# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
      -f2_grad_x1 = grad(f2,0)
      -
      -# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
      -f2_grad_x2 = grad(f2,1)
      -
      -x1 = 1.0
      -x2 = 3.0 
      -
      -print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
      -print("-"*30)
      -
      -# Compare with the analytical derivatives:
      -
      -# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
      -f2_grad_x1_analytical = 9*x1**2 + x2
      -
      -# Derivative of f2 w.r.t x2 is: x1 - 5:
      -f2_grad_x2_analytical = x1 - 5
      -
      -# See the evaluated derivations:
      -print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
      -print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
      -
      -print()
      -
      -print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
      -print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
      -
      -

      -Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable. - -

      -









      - -

      More complicated functions using the elements of their arguments directly

      - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f3(x): # Assumes x is an array of length 5 or higher
      -    return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
      -
      -f3_grad = grad(f3)
      -
      -x = np.linspace(0,4,5)
      -
      -# Print the computed gradient:
      -print("The computed gradient of f3 is: ", f3_grad(x))
      -
      -# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
      -f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
      -
      -# Print the analytical gradient:
      -print("The analytical gradient of f3 is: ", f3_grad_analytical)
      -
      -

      -Note that in this case, when sending an array as input argument, the -output from Autograd is another array. This is the true gradient of -the function, as opposed to the function in the previous example. By -using arrays to represent the variables, the output from Autograd -might be easier to work with, as the output is closer to what one -could expect form a gradient-evaluting function. - -

      - - -

      Functions using mathematical functions from Numpy

      - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f4(x):
      -    return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
      -
      -f4_grad = grad(f4)
      -
      -x = 2.7
      -
      -# Print the computed derivative:
      -print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
      -
      -# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
      -f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
      -
      -# Print the analytical gradient:
      -print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
      -
      -

      -









      - -

      More autograd

      - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f5(x):
      -    if x >= 0:
      -        return x**2
      -    else:
      -        return -3*x + 1
      -
      -f5_grad = grad(f5)
      -
      -x = 2.7
      -
      -# Print the computed derivative:
      -print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
      -
      -

      -









      - -

      And with loops

      - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f6_for(x):
      -    val = 0
      -    for i in range(10):
      -        val = val + x**i
      -    return val
      -
      -def f6_while(x):
      -    val = 0
      -    i = 0
      -    while i < 10:
      -        val = val + x**i
      -        i = i + 1
      -    return val
      -
      -f6_for_grad = grad(f6_for)
      -f6_while_grad = grad(f6_while)
      -
      -x = 0.5
      -
      -# Print the computed derivaties of f6_for and f6_while
      -print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
      -print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
      -
      -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
      -# The analytical derivative is: sum(i*x**(i-1)) 
      -f6_grad_analytical = 0
      -for i in range(10):
      -    f6_grad_analytical += i*x**(i-1)
      -
      -print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
      -
      -

      -









      - -

      Using recursion

      -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -
      -def f7(n): # Assume that n is an integer
      -    if n == 1 or n == 0:
      -        return 1
      -    else:
      -        return n*f7(n-1)
      -
      -f7_grad = grad(f7)
      -
      -n = 2.0
      -
      -print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
      -
      -# The function f7 is an implementation of the factorial of n.
      -# By using the product rule, one can find that the derivative is:
      -
      -f7_grad_analytical = 0
      -for i in range(int(n)-1):
      -    tmp = 1
      -    for k in range(int(n)-1):
      -        if k != i:
      -            tmp *= (n - k)
      -    f7_grad_analytical += tmp
      -
      -print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
      -
      -

      -Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input. - -

      -









      - -

      Unsupported functions

      -Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd. - -

      -Assigning a value to the variable being differentiated with respect to -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f8(x): # Assume x is an array
      -    x[2] = 3
      -    return x*2
      -
      -f8_grad = grad(f8)
      -
      -x = 8.4
      -
      -print("The derivative of f8 is:",f8_grad(x))
      -
      -

      -Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible. - -

      -









      - -

      The syntax a.dot(b) when finding the dot product

      -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f9(a): # Assume a is an array with 2 elements
      -    b = np.array([1.0,2.0])
      -    return a.dot(b)
      -
      -f9_grad = grad(f9)
      -
      -x = np.array([1.0,0.0])
      -
      -print("The derivative of f9 is:",f9_grad(x))
      -
      -

      -Here we are told that the 'dot' function does not belong to Autograd's -version of a Numpy array. To overcome this, an alternative syntax -which also computed the dot product can be used: - -

      - - -

      import autograd.numpy as np
      -from autograd import grad
      -def f9_alternative(x): # Assume a is an array with 2 elements
      -    b = np.array([1.0,2.0])
      -    return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
      -
      -f9_alternative_grad = grad(f9_alternative)
      -
      -x = np.array([3.0,0.0])
      -
      -print("The gradient of f9 is:",f9_alternative_grad(x))
      -
      -# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
      -# w.r.t x is (b_1, b_2).
      -
      -

      -









      - -

      Recommended to avoid

      -The documentation recommends to avoid inplace operations such as -

      - - -

      a += b
      -a -= b
      -a*= b
      -a /=b
      -
      -

      -









      - -

      Stochastic Gradient Descent

      +

      Stochastic Gradient Descent

      Stochastic gradient descent (SGD) and variants thereof address some of @@ -1918,7 +926,7 @@ $$











      -

      Computation of gradients

      +

      Computation of gradients

      This in turn means that the gradient can be @@ -1938,7 +946,7 @@ minibatches. We denote these minibatches by \( B_k \) where











      -

      SGD example

      +

      SGD example

      As an example, suppose we have \( 10 \) data points \( (\mathbf{x}_1,\cdots, \mathbf{x}_{10}) \) and we choose to have \( M=5 \) minibathces, then each minibatch contains two data points. In particular we have @@ -1962,7 +970,7 @@ $$











      -

      The gradient step

      +

      The gradient step

      Thus a gradient descent step now looks like @@ -1981,7 +989,7 @@ the number of minibatches, as exemplified in the code below.











      -

      Simple example code

      +

      Simple example code

      @@ -2013,7 +1021,7 @@ all \( n \) datapoints.











      -

      When do we stop?

      +

      When do we stop?

      A natural question is when do we stop the search for a new minimum? @@ -2030,7 +1038,7 @@ gave the lowest value.











      -

      Slightly different approach

      +

      Slightly different approach

      Another approach is to let the step length \( \gamma_j \) depend on the @@ -2078,7 +1086,7 @@ j = 0











      -

      Program for stochastic gradient

      +

      Program for stochastic gradient

      @@ -2157,7 +1165,7 @@ plt.show()











      -

      Using gradient descent methods, limitations

      +

      Using gradient descent methods, limitations

      • Gradient descent (GD) finds local minima of our function. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our energy function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.
      • @@ -2170,7 +1178,7 @@ plt.show()









        -

        Momentum based GD

        +

        Momentum based GD

        The stochastic gradient descent (SGD) is almost always used with a momentum or inertia term that serves as a memory of the direction we are moving in parameter space. This is typically @@ -2193,7 +1201,7 @@ where we have defined \( \Delta \boldsymbol{\theta}_{t}= \boldsymbol{\theta}_t-\











        -

        More on momentum based approaches

        +

        More on momentum based approaches

        Let us try to get more intuition from these equations. It is helpful to consider a simple physical analogy with a particle of mass \( m \) moving in a viscous medium with drag coefficient \( \mu \) and potential @@ -2215,7 +1223,7 @@ $$











        -

        Momentum parameter

        +

        Momentum parameter

        Notice that this equation is identical to previous one if we identify the position of the particle, \( \mathbf{w} \), with the parameters \( \boldsymbol{\theta} \). This allows us to identify the momentum parameter and learning rate with the mass of the particle and the viscous drag as: $$ @@ -2245,7 +1253,7 @@ One of the major advantages of NAG is that it allows for the use of a larger lea











        -

        Second moment of the gradient

        +

        Second moment of the gradient

        In stochastic gradient descent, with and without momentum, we still @@ -2270,7 +1278,7 @@ Recently, a number of methods have been introduced that accomplish this by track











        -

        RMS prop

        +

        RMS prop

        In RMS prop, in addition to keeping a running average of the first moment of the gradient, we also keep track of the second moment denoted by \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}_t^2] \). The update rule for RMS prop is given by @@ -2288,7 +1296,7 @@ where \( \beta \) controls the averaging time of the second moment and is typica











        -

        ADAM optimizer

        +

        ADAM optimizer

        A related algorithm is the ADAM optimizer. In ADAM, we keep a running average of both the first and second moment of the gradient and use this information to adaptively change the learning rate for different parameters. In addition to keeping a running average of the first and second moments of the gradient (i.e. \( \mathbf{m}_t=\mathbb{E}[\mathbf{g}_t] \) and \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}^2_t] \), respectively), ADAM performs an additional bias correction to account for the fact that we are estimating the first two moments of the gradient using a running average (denoted by the hats in the update rule below). The update rule for ADAM is given by (where multiplication and division are once again understood to be element-wise operations below) @@ -2316,7 +1324,7 @@ $$











        -

        Practical tips

        +

        Practical tips

        • Randomize the data when making mini-batches. It is always important to randomly shuffle the data when forming mini-batches. Otherwise, the gradient descent method can fit spurious correlations resulting from the order in which data is presented.
        • @@ -2327,11 +1335,1012 @@ $$ Geron's text, see chapter 11, has several interesting discussions. +

          +









          + +

          Automatic differentiation

          +Python has tools for so-called automatic differentiation. +Consider the following example +$$ +f(x) = \sin\left(2\pi x + x^2\right) +$$ + +which has the following derivative +$$ +f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) +$$ + +Using autograd we have + +

          + + +

          import autograd.numpy as np
          +
          +# To do elementwise differentiation:
          +from autograd import elementwise_grad as egrad 
          +
          +# To plot:
          +import matplotlib.pyplot as plt 
          +
          +
          +def f(x):
          +    return np.sin(2*np.pi*x + x**2)
          +
          +def f_grad_analytic(x):
          +    return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
          +
          +# Do the comparison:
          +x = np.linspace(0,1,1000)
          +
          +f_grad = egrad(f)
          +
          +computed = f_grad(x)
          +analytic = f_grad_analytic(x)
          +
          +plt.title('Derivative computed from Autograd compared with the analytical derivative')
          +plt.plot(x,computed,label='autograd')
          +plt.plot(x,analytic,label='analytic')
          +
          +plt.xlabel('x')
          +plt.ylabel('y')
          +plt.legend()
          +
          +plt.show()
          +
          +print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
          +
          +

          + + +

          Using autograd

          + +

          +Here we +experiment with what kind of functions Autograd is capable +of finding the gradient of. The following Python functions are just +meant to illustrate what Autograd can do, but please feel free to +experiment with other, possibly more complicated, functions as well. + +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +
          +def f1(x):
          +    return x**3 + 1
          +
          +f1_grad = grad(f1)
          +
          +# Remember to send in float as argument to the computed gradient from Autograd!
          +a = 1.0
          +
          +# See the evaluated gradient at a using autograd:
          +print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
          +
          +# Compare with the analytical derivative, that is f1'(x) = 3*x**2 
          +grad_analytical = 3*a**2
          +print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
          +
          +

          +









          + +

          Autograd with more complicated functions

          + +

          +To differentiate with respect to two (or more) arguments of a Python +function, Autograd need to know at which variable the function if +being differentiated with respect to. + +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +def f2(x1,x2):
          +    return 3*x1**3 + x2*(x1 - 5) + 1
          +
          +# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
          +f2_grad_x1 = grad(f2,0)
          +
          +# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
          +f2_grad_x2 = grad(f2,1)
          +
          +x1 = 1.0
          +x2 = 3.0 
          +
          +print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
          +print("-"*30)
          +
          +# Compare with the analytical derivatives:
          +
          +# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
          +f2_grad_x1_analytical = 9*x1**2 + x2
          +
          +# Derivative of f2 w.r.t x2 is: x1 - 5:
          +f2_grad_x2_analytical = x1 - 5
          +
          +# See the evaluated derivations:
          +print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
          +print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
          +
          +print()
          +
          +print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
          +print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
          +
          +

          +Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable. + +

          +









          + +

          More complicated functions using the elements of their arguments directly

          + +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +def f3(x): # Assumes x is an array of length 5 or higher
          +    return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
          +
          +f3_grad = grad(f3)
          +
          +x = np.linspace(0,4,5)
          +
          +# Print the computed gradient:
          +print("The computed gradient of f3 is: ", f3_grad(x))
          +
          +# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
          +f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
          +
          +# Print the analytical gradient:
          +print("The analytical gradient of f3 is: ", f3_grad_analytical)
          +
          +

          +Note that in this case, when sending an array as input argument, the +output from Autograd is another array. This is the true gradient of +the function, as opposed to the function in the previous example. By +using arrays to represent the variables, the output from Autograd +might be easier to work with, as the output is closer to what one +could expect form a gradient-evaluting function. + +

          + + +

          Functions using mathematical functions from Numpy

          + +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +def f4(x):
          +    return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
          +
          +f4_grad = grad(f4)
          +
          +x = 2.7
          +
          +# Print the computed derivative:
          +print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
          +
          +# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
          +f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
          +
          +# Print the analytical gradient:
          +print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
          +
          +

          +









          + +

          More autograd

          + +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +def f5(x):
          +    if x >= 0:
          +        return x**2
          +    else:
          +        return -3*x + 1
          +
          +f5_grad = grad(f5)
          +
          +x = 2.7
          +
          +# Print the computed derivative:
          +print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
          +
          +

          +









          + +

          And with loops

          + +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +def f6_for(x):
          +    val = 0
          +    for i in range(10):
          +        val = val + x**i
          +    return val
          +
          +def f6_while(x):
          +    val = 0
          +    i = 0
          +    while i < 10:
          +        val = val + x**i
          +        i = i + 1
          +    return val
          +
          +f6_for_grad = grad(f6_for)
          +f6_while_grad = grad(f6_while)
          +
          +x = 0.5
          +
          +# Print the computed derivaties of f6_for and f6_while
          +print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
          +print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
          +
          +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
          +# The analytical derivative is: sum(i*x**(i-1)) 
          +f6_grad_analytical = 0
          +for i in range(10):
          +    f6_grad_analytical += i*x**(i-1)
          +
          +print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
          +
          +

          +









          + +

          Using recursion

          +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +
          +def f7(n): # Assume that n is an integer
          +    if n == 1 or n == 0:
          +        return 1
          +    else:
          +        return n*f7(n-1)
          +
          +f7_grad = grad(f7)
          +
          +n = 2.0
          +
          +print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
          +
          +# The function f7 is an implementation of the factorial of n.
          +# By using the product rule, one can find that the derivative is:
          +
          +f7_grad_analytical = 0
          +for i in range(int(n)-1):
          +    tmp = 1
          +    for k in range(int(n)-1):
          +        if k != i:
          +            tmp *= (n - k)
          +    f7_grad_analytical += tmp
          +
          +print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
          +
          +

          +Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input. + +

          +









          + +

          Unsupported functions

          +Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd. + +

          +Assigning a value to the variable being differentiated with respect to +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +def f8(x): # Assume x is an array
          +    x[2] = 3
          +    return x*2
          +
          +f8_grad = grad(f8)
          +
          +x = 8.4
          +
          +print("The derivative of f8 is:",f8_grad(x))
          +
          +

          +Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible. + +

          +









          + +

          The syntax a.dot(b) when finding the dot product

          +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +def f9(a): # Assume a is an array with 2 elements
          +    b = np.array([1.0,2.0])
          +    return a.dot(b)
          +
          +f9_grad = grad(f9)
          +
          +x = np.array([1.0,0.0])
          +
          +print("The derivative of f9 is:",f9_grad(x))
          +
          +

          +Here we are told that the 'dot' function does not belong to Autograd's +version of a Numpy array. To overcome this, an alternative syntax +which also computed the dot product can be used: + +

          + + +

          import autograd.numpy as np
          +from autograd import grad
          +def f9_alternative(x): # Assume a is an array with 2 elements
          +    b = np.array([1.0,2.0])
          +    return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
          +
          +f9_alternative_grad = grad(f9_alternative)
          +
          +x = np.array([3.0,0.0])
          +
          +print("The gradient of f9 is:",f9_alternative_grad(x))
          +
          +# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
          +# w.r.t x is (b_1, b_2).
          +
          +

          +









          + +

          Recommended to avoid

          +The documentation recommends to avoid inplace operations such as +

          + + +

          a += b
          +a -= b
          +a*= b
          +a /=b
          +
          +

          +









          + +

          Standard steepest descent

          + +

          +Before we proceed, we would like to discuss the approach called the +standard Steepest descent, which again leads to us having to be able +to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG). + +

          +The success of the CG method +for finding solutions of non-linear problems is based on the theory +of conjugate gradients for linear systems of equations. It belongs to +the class of iterative methods for solving problems from linear +algebra of the type +$$ +\begin{equation*} +\hat{A}\hat{x} = \hat{b}. +\end{equation*} +$$ + +

          +In the iterative process we end up with a problem like + +$$ +\begin{equation*} + \hat{r}= \hat{b}-\hat{A}\hat{x}, +\end{equation*} +$$ + +where \( \hat{r} \) is the so-called residual or error in the iterative process. + +

          +When we have found the exact solution, \( \hat{r}=0 \). + +

          +









          + +

          Gradient method

          + +

          +The residual is zero when we reach the minimum of the quadratic equation +$$ +\begin{equation*} + P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b}, +\end{equation*} +$$ + +

          +with the constraint that the matrix \( \hat{A} \) is positive definite and +symmetric. This defines also the Hessian and we want it to be positive definite. + +

          +









          + +

          Steepest descent method

          + +

          +We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). +We can assume without loss of generality that +$$ +\begin{equation*} +\hat{x}_0=0, +\end{equation*} +$$ + +or consider the system +$$ +\begin{equation*} +\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, +\end{equation*} +$$ + +instead. + +

          +









          + +

          Steepest descent method

          +
          + +

          +One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form +$$ +\begin{equation*} + f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. +\end{equation*} +$$ + +This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition) +to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), +which equals +$$ +\begin{equation*} +\hat{A}\hat{x}_0-\hat{b}, +\end{equation*} +$$ + +and +\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). + + +

          + + +

          +









          + +

          Final expressions

          +
          + +

          +We can compute the residual iteratively as +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, + \end{equation*} +$$ + +which equals +$$ +\begin{equation*} +\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k), + \end{equation*} +$$ + +or +$$ +\begin{equation*} +(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k, + \end{equation*} +$$ + +which gives + +$$ +\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k} +$$ + +leading to the iterative scheme +$$ +\begin{equation*} +\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k}, + \end{equation*} +$$ +

          + + +

          +









          + +

          Code examples for steepest descent

          + +

          +









          + +

          Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come

          +
          + +

          +

          + + +

          #include <cmath>
          +#include <iostream>
          +#include <fstream>
          +#include <iomanip>
          +#include "vectormatrixclass.h"
          +using namespace  std;
          +//   Main function begins here
          +int main(int  argc, char * argv[]){
          +  int dim = 2;
          +  Vector x(dim),xsd(dim), b(dim),x0(dim);
          +  Matrix A(dim,dim);
          +
          +  // Set our initial guess
          +  x0(0) = x0(1) = 0;
          +  // Set the matrix
          +  A(0,0) =  3;    A(1,0) =  2;   A(0,1) =  2;   A(1,1) =  6;
          +  b(0) = 2; b(1) = -8;
          +  cout << "The Matrix A that we are using: " << endl;
          +  A.Print();
          +  cout << endl;
          +  xsd = SteepestDescent(A,b,x0);
          +  cout << "The approximate solution using Steepest Descent is: " << endl;
          +  xsd.Print();
          +  cout << endl;
          +}
          +
          + +
          + + +

          +









          + +

          The routine for the steepest descent method

          +
          + +

          +

          + + +

          Vector SteepestDescent(Matrix A, Vector b, Vector x0){
          +  int IterMax, i;
          +  int dim = x0.Dimension();
          +  const double tolerance = 1.0e-14;
          +  Vector x(dim),f(dim),z(dim);
          +  double c,alpha,d;
          +  IterMax = 30;
          +  x = x0;
          +  r = A*x-b;
          +  i = 0;
          +  while (i <= IterMax){
          +    z = A*r;
          +    c = dot(r,r);
          +    alpha = c/dot(r,z);
          +    x = x - alpha*r;
          +    r =  A*x-b;
          +    if(sqrt(dot(r,r)) < tolerance) break;
          +    i++;
          +  }
          +  return x;
          +}
          +
          + +
          + + +

          +









          + +

          Steepest descent example

          + +

          + + +

          import numpy as np
          +import numpy.linalg as la
          +
          +import scipy.optimize as sopt
          +
          +import matplotlib.pyplot as pt
          +from mpl_toolkits.mplot3d import axes3d
          +
          +def f(x):
          +    return 0.5*x[0]**2 + 2.5*x[1]**2
          +
          +def df(x):
          +    return np.array([x[0], 5*x[1]])
          +
          +fig = pt.figure()
          +ax = fig.gca(projection="3d")
          +
          +xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]
          +fmesh = f(np.array([xmesh, ymesh]))
          +ax.plot_surface(xmesh, ymesh, fmesh)
          +
          +

          +And then as countor plot +

          + + +

          pt.axis("equal")
          +pt.contour(xmesh, ymesh, fmesh)
          +guesses = [np.array([2, 2./5])]
          +
          +

          +Find guesses +

          + + +

          x = guesses[-1]
          +s = -df(x)
          +
          +

          +Run it! +

          + + +

          def f1d(alpha):
          +    return f(x + alpha*s)
          +
          +alpha_opt = sopt.golden(f1d)
          +next_guess = x + alpha_opt * s
          +guesses.append(next_guess)
          +print(next_guess)
          +
          +

          +What happened? +

          + + +

          pt.axis("equal")
          +pt.contour(xmesh, ymesh, fmesh, 50)
          +it_array = np.array(guesses)
          +pt.plot(it_array.T[0], it_array.T[1], "x-")
          +
          +

          +









          + +

          Conjugate gradient method

          +
          + +

          +In the CG method we define so-called conjugate directions and two vectors +\( \hat{s} \) and \( \hat{t} \) +are said to be +conjugate if +$$ +\begin{equation*} +\hat{s}^T\hat{A}\hat{t}= 0. +\end{equation*} +$$ + +The philosophy of the CG method is to perform searches in various conjugate directions +of our vectors \( \hat{x}_i \) obeying the above criterion, namely +$$ +\begin{equation*} +\hat{x}_i^T\hat{A}\hat{x}_j= 0. +\end{equation*} +$$ + +Two vectors are conjugate if they are orthogonal with respect to +this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \). +

          + + +

          +









          + +

          Conjugate gradient method

          +
          + +

          +An example is given by the eigenvectors of the matrix +$$ +\begin{equation*} +\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j, +\end{equation*} +$$ + +which is zero unless \( i=j \). +

          + + +

          +









          + +

          Conjugate gradient method

          +
          + +

          +Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size +\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector +$$ +\begin{equation*} +\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}. +\end{equation*} +$$ + +We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions. +Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution +$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely + +$$ +\begin{equation*} + \hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i. +\end{equation*} +$$ +

          + + +

          +









          + +

          Conjugate gradient method

          +
          + +

          +The coefficients are given by +$$ +\begin{equation*} + \mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. +\end{equation*} +$$ + +Multiplying with \( \hat{p}_k^T \) from the left gives + +$$ +\begin{equation*} + \hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b}, +\end{equation*} +$$ + +and we can define the coefficients \( \alpha_k \) as + +$$ +\begin{equation*} + \alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k} +\end{equation*} +$$ +

          + + +

          +









          + +

          Conjugate gradient method and iterations

          +
          + +

          + +

          +If we choose the conjugate vectors \( \hat{p}_k \) carefully, +then we may not need all of them to obtain a good approximation to the solution +\( \hat{x} \). +We want to regard the conjugate gradient method as an iterative method. +This will us to solve systems where \( n \) is so large that the direct +method would take too much time. + +

          +We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). +We can assume without loss of generality that +$$ +\begin{equation*} +\hat{x}_0=0, +\end{equation*} +$$ + +or consider the system +$$ +\begin{equation*} +\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, +\end{equation*} +$$ + +instead. +

          + + +

          +









          + +

          Conjugate gradient method

          +
          + +

          +One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form +$$ +\begin{equation*} + f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. +\end{equation*} +$$ + +This suggests taking the first basis vector \( \hat{p}_1 \) +to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), +which equals +$$ +\begin{equation*} +\hat{A}\hat{x}_0-\hat{b}, +\end{equation*} +$$ + +and +\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). +The other vectors in the basis will be conjugate to the gradient, +hence the name conjugate gradient method. +

          + + +

          +









          + +

          Conjugate gradient method

          +
          + +

          +Let \( \hat{r}_k \) be the residual at the \( k \)-th step: +$$ +\begin{equation*} +\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k. +\end{equation*} +$$ + +Note that \( \hat{r}_k \) is the negative gradient of \( f \) at +\( \hat{x}=\hat{x}_k \), +so the gradient descent method would be to move in the direction \( \hat{r}_k \). +Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other, +so we take the direction closest to the gradient \( \hat{r}_k \) +under the conjugacy constraint. +This gives the following expression +$$ +\begin{equation*} +\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k. +\end{equation*} +$$ +

          + + +

          +









          + +

          Conjugate gradient method

          +
          + +

          +We can also compute the residual iteratively as +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, + \end{equation*} +$$ + +which equals +$$ +\begin{equation*} +\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k), + \end{equation*} +$$ + +or +$$ +\begin{equation*} +(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k, + \end{equation*} +$$ + +which gives + +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k}, + \end{equation*} +$$ +

          + + +

          +









          + +

          Simple implementation of the Conjugate gradient algorithm

          +
          + +

          +

          + + +

            Vector ConjugateGradient(Matrix A, Vector b, Vector x0){
          +  int dim = x0.Dimension();
          +  const double tolerance = 1.0e-14;
          +  Vector x(dim),r(dim),v(dim),z(dim);
          +  double c,t,d;
          +
          +  x = x0;
          +  r = b - A*x;
          +  v = r;
          +  c = dot(r,r);
          +  int i = 0; IterMax = dim;
          +  while(i <= IterMax){
          +    z = A*v;
          +    t = c/dot(v,z);
          +    x = x + t*v;
          +    r = r - t*z;
          +    d = dot(r,r);
          +    if(sqrt(d) < tolerance)
          +      break;
          +    v = r + (d/c)*v;
          +    c = d;  i++;
          +  }
          +  return x;
          +} 
          +
          + +
          + + +

          +









          + +

          Broyden–Fletcher–Goldfarb–Shanno algorithm

          +
          + +

          +The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take. + +

          +The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage. + +

          +The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation +$$ +B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}), +$$ + +

          +where \( B_{k} \) is an approximation to the Hessian matrix, which is +updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \) +is the gradient of the function +evaluated at \( x_k \). +A line search in the direction \( p_k \) is then used to +find the next point \( x_{k+1} \) by minimising +$$ +f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}), +$$ + +over the scalar \( \alpha > 0 \). + + +

          + + +

          +

          - © 1999-2018, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license + © 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
          diff --git a/doc/pub/Splines/html/Splines.html b/doc/pub/Splines/html/Splines.html index f8973a018..299f5c123 100644 --- a/doc/pub/Splines/html/Splines.html +++ b/doc/pub/Splines/html/Splines.html @@ -6,6 +6,7 @@ Automatically generated HTML file from DocOnce source + Data Analysis and Machine Learning Lectures: Optimization and Gradient Methods @@ -65,113 +66,114 @@ div { text-align: justify; text-justify: inter-word; } @@ -213,12 +215,17 @@ MathJax.Hub.Config({
          [2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

          -

          Oct 18, 2018

          +

          Sep 19, 2019












          -

          Optimization, the central part of any Machine Learning algortithm

          +

          Optimization problems, why?

          + +

          +









          + +

          Optimization, the central part of any Machine Learning algortithm

          Almost every problem in machine learning and data science starts with @@ -233,7 +240,7 @@ some approximative/numerical method to compute the minimum.











          -

          Revisiting our Logistic Regression case

          +

          Revisiting our Logistic Regression case

          In our discussion on Logistic Regression we studied the @@ -255,7 +262,7 @@ where \( \hat{\beta} \) are the weights we wish to extract from data, in our cas











          -

          The equations to solve

          +

          The equations to solve

          Our compact equations used a definition of a vector \( \hat{y} \) with \( n \) @@ -281,7 +288,7 @@ This defines what is called the Hessian matrix.











          -

          Solving using Newton-Raphson's method

          +

          Solving using Newton-Raphson's method

          If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives. @@ -307,7 +314,7 @@ If we can compute these matrices, in particular the Hessian, the above is often











          -

          Brief reminder on Newton-Raphson's method

          +

          Brief reminder on Newton-Raphson's method

          Let us quickly remind ourselves how we derive the above method. @@ -324,7 +331,7 @@ normally discourage the use of this method.











          -

          The equations

          +

          The equations

          The Newton-Raphson formula consists geometrically of extending the @@ -360,7 +367,7 @@ $$











          -

          Simple geometric interpretation

          +

          Simple geometric interpretation

          The above is Newton-Raphson's method. It has a simple geometric @@ -378,7 +385,7 @@ vanishes, then Newton-Raphson may fail totally











          -

          Extending to more than one variable

          +

          Extending to more than one variable

          Newton's method can be generalized to systems of several non-linear equations @@ -433,7 +440,7 @@ more than two non-linear equations. In our case, the Jacobian matrix is given by











          -

          Steepest descent

          +

          Steepest descent

          The basic idea of gradient descent is @@ -457,7 +464,7 @@ we are always moving towards smaller function values, i.e a minimum.

          -

          More on Steepest descent

          +

          More on Steepest descent

          The previous observation is the basis of the method of steepest @@ -476,7 +483,7 @@ the learning rate within the context of Machine Learning.

          -

          The ideal

          +

          The ideal

          Ideally the sequence \( \{\mathbf{x}_k \}_{k=0} \) converges to a global @@ -502,7 +509,7 @@ Note that the gradient is a function of \( \mathbf{x} =

          -

          The sensitiveness of the gradient descent

          +

          The sensitiveness of the gradient descent

          The gradient descent method @@ -521,7 +528,7 @@ randomness. One such method is that of Stochastic Gradient Descent

          -

          Convex functions

          +

          Convex functions

          Ideally we want our cost/loss function to be convex(concave). @@ -541,7 +548,7 @@ regular polygons (triangles, rectangles, pentagons, etc...).











          -

          Convex function

          +

          Convex function

          Convex function: Let \( X \subset \mathbb{R}^n \) be a convex set. Assume that the function \( f: X \rightarrow \mathbb{R} \) is continuous, then \( f \) is said to be convex if $$f(tx_1 + (1-t)x_2) \leq tf(x_1) + (1-t)f(x_2) $$ for all \( x_1, x_2 \in X \) and for all \( t \in [0,1] \). If \( \leq \) is replaced with a strict inequaltiy in the definition, we demand \( x_1 \neq x_2 \) and \( t\in(0,1) \) then \( f \) is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting \( f(x_1) \) and \( f(x_2) \), the value of the function on the interval \( [x_1,x_2] \) is always below the line as illustrated below. @@ -549,7 +556,7 @@ regular polygons (triangles, rectangles, pentagons, etc...).











          -

          Conditions on convex functions

          +

          Conditions on convex functions

          In the following we state first and second-order conditions which @@ -591,7 +598,7 @@ This condition is particularly useful since it gives us an procedure for determi











          -

          More on convex functions

          +

          More on convex functions

          The next result is of great importance to us and the reason why we are @@ -621,7 +628,7 @@ This result means that if we know that the cost/loss function is convex and we a











          -

          Some simple problems

          +

          Some simple problems

          1. Show that \( f(x)=x^2 \) is convex for \( x \in \mathbb{R} \) using the definition of convexity. Hint: If you re-write the definition, \( f \) is convex if the following holds for all \( x,y \in D_f \) and any \( \lambda \in [0,1] \) $\lambda f(x)+(1-\lambda)f(y)-f(\lambda x + (1-\lambda) y ) \geq 0$.
          2. @@ -645,623 +652,10 @@ This result means that if we know that the cost/loss function is convex and we a Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this). -

            -









            - -

            Standard steepest descent

            - -

            -Before we proceed, we would like to discuss the approach called the -standard Steepest descent, which again leads to us having to be able -to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG). - -

            -The success of the CG method -for finding solutions of non-linear problems is based on the theory -of conjugate gradients for linear systems of equations. It belongs to -the class of iterative methods for solving problems from linear -algebra of the type -$$ -\begin{equation*} -\hat{A}\hat{x} = \hat{b}. -\end{equation*} -$$ - -

            -In the iterative process we end up with a problem like - -$$ -\begin{equation*} - \hat{r}= \hat{b}-\hat{A}\hat{x}, -\end{equation*} -$$ - -where \( \hat{r} \) is the so-called residual or error in the iterative process. - -

            -When we have found the exact solution, \( \hat{r}=0 \). - -

            -









            - -

            Gradient method

            - -

            -The residual is zero when we reach the minimum of the quadratic equation -$$ -\begin{equation*} - P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b}, -\end{equation*} -$$ - -

            -with the constraint that the matrix \( \hat{A} \) is positive definite and -symmetric. This defines also the Hessian and we want it to be positive definite. - -

            -









            - -

            Steepest descent method

            - -

            -We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). -We can assume without loss of generality that -$$ -\begin{equation*} -\hat{x}_0=0, -\end{equation*} -$$ - -or consider the system -$$ -\begin{equation*} -\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, -\end{equation*} -$$ - -instead. - -

            -









            - -

            Steepest descent method

            -
            - -

            -One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form -$$ -\begin{equation*} - f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. -\end{equation*} -$$ - -This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition) -to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), -which equals -$$ -\begin{equation*} -\hat{A}\hat{x}_0-\hat{b}, -\end{equation*} -$$ - -and -\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). - - -

            - - -

            -









            - -

            Final expressions

            -
            - -

            -We can compute the residual iteratively as -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, - \end{equation*} -$$ - -which equals -$$ -\begin{equation*} -\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k), - \end{equation*} -$$ - -or -$$ -\begin{equation*} -(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k, - \end{equation*} -$$ - -which gives - -$$ -\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k} -$$ - -leading to the iterative scheme -$$ -\begin{equation*} -\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k}, - \end{equation*} -$$ -

            - - -

            -









            - -

            Code examples for steepest descent

            - -

            -









            - -

            Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come

            -
            - -

            -

            - - -

            #include <cmath>
            -#include <iostream>
            -#include <fstream>
            -#include <iomanip>
            -#include "vectormatrixclass.h"
            -using namespace  std;
            -//   Main function begins here
            -int main(int  argc, char * argv[]){
            -  int dim = 2;
            -  Vector x(dim),xsd(dim), b(dim),x0(dim);
            -  Matrix A(dim,dim);
            -
            -  // Set our initial guess
            -  x0(0) = x0(1) = 0;
            -  // Set the matrix
            -  A(0,0) =  3;    A(1,0) =  2;   A(0,1) =  2;   A(1,1) =  6;
            -  b(0) = 2; b(1) = -8;
            -  cout << "The Matrix A that we are using: " << endl;
            -  A.Print();
            -  cout << endl;
            -  xsd = SteepestDescent(A,b,x0);
            -  cout << "The approximate solution using Steepest Descent is: " << endl;
            -  xsd.Print();
            -  cout << endl;
            -}
            -
            - -
            - - -

            -









            - -

            The routine for the steepest descent method

            -
            - -

            -

            - - -

            Vector SteepestDescent(Matrix A, Vector b, Vector x0){
            -  int IterMax, i;
            -  int dim = x0.Dimension();
            -  const double tolerance = 1.0e-14;
            -  Vector x(dim),f(dim),z(dim);
            -  double c,alpha,d;
            -  IterMax = 30;
            -  x = x0;
            -  r = A*x-b;
            -  i = 0;
            -  while (i <= IterMax){
            -    z = A*r;
            -    c = dot(r,r);
            -    alpha = c/dot(r,z);
            -    x = x - alpha*r;
            -    r =  A*x-b;
            -    if(sqrt(dot(r,r)) < tolerance) break;
            -    i++;
            -  }
            -  return x;
            -}
            -
            - -
            - - -

            -









            - -

            Steepest descent example

            - -

            - - -

            import numpy as np
            -import numpy.linalg as la
            -
            -import scipy.optimize as sopt
            -
            -import matplotlib.pyplot as pt
            -from mpl_toolkits.mplot3d import axes3d
            -
            -def f(x):
            -    return 0.5*x[0]**2 + 2.5*x[1]**2
            -
            -def df(x):
            -    return np.array([x[0], 5*x[1]])
            -
            -fig = pt.figure()
            -ax = fig.gca(projection="3d")
            -
            -xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]
            -fmesh = f(np.array([xmesh, ymesh]))
            -ax.plot_surface(xmesh, ymesh, fmesh)
            -
            -

            -And then as countor plot -

            - - -

            pt.axis("equal")
            -pt.contour(xmesh, ymesh, fmesh)
            -guesses = [np.array([2, 2./5])]
            -
            -

            -Find guesses -

            - - -

            x = guesses[-1]
            -s = -df(x)
            -
            -

            -Run it! -

            - - -

            def f1d(alpha):
            -    return f(x + alpha*s)
            -
            -alpha_opt = sopt.golden(f1d)
            -next_guess = x + alpha_opt * s
            -guesses.append(next_guess)
            -print(next_guess)
            -
            -

            -What happened? -

            - - -

            pt.axis("equal")
            -pt.contour(xmesh, ymesh, fmesh, 50)
            -it_array = np.array(guesses)
            -pt.plot(it_array.T[0], it_array.T[1], "x-")
            -
            -

            -









            - -

            Conjugate gradient method

            -
            - -

            -In the CG method we define so-called conjugate directions and two vectors -\( \hat{s} \) and \( \hat{t} \) -are said to be -conjugate if -$$ -\begin{equation*} -\hat{s}^T\hat{A}\hat{t}= 0. -\end{equation*} -$$ - -The philosophy of the CG method is to perform searches in various conjugate directions -of our vectors \( \hat{x}_i \) obeying the above criterion, namely -$$ -\begin{equation*} -\hat{x}_i^T\hat{A}\hat{x}_j= 0. -\end{equation*} -$$ - -Two vectors are conjugate if they are orthogonal with respect to -this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \). -

            - - -

            -









            - -

            Conjugate gradient method

            -
            - -

            -An example is given by the eigenvectors of the matrix -$$ -\begin{equation*} -\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j, -\end{equation*} -$$ - -which is zero unless \( i=j \). -

            - - -

            -









            - -

            Conjugate gradient method

            -
            - -

            -Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size -\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector -$$ -\begin{equation*} -\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}. -\end{equation*} -$$ - -We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions. -Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution -$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely - -$$ -\begin{equation*} - \hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i. -\end{equation*} -$$ -

            - - -

            -









            - -

            Conjugate gradient method

            -
            - -

            -The coefficients are given by -$$ -\begin{equation*} - \mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. -\end{equation*} -$$ - -Multiplying with \( \hat{p}_k^T \) from the left gives - -$$ -\begin{equation*} - \hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b}, -\end{equation*} -$$ - -and we can define the coefficients \( \alpha_k \) as - -$$ -\begin{equation*} - \alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k} -\end{equation*} -$$ -

            - - -

            -









            - -

            Conjugate gradient method and iterations

            -
            - -

            - -

            -If we choose the conjugate vectors \( \hat{p}_k \) carefully, -then we may not need all of them to obtain a good approximation to the solution -\( \hat{x} \). -We want to regard the conjugate gradient method as an iterative method. -This will us to solve systems where \( n \) is so large that the direct -method would take too much time. - -

            -We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). -We can assume without loss of generality that -$$ -\begin{equation*} -\hat{x}_0=0, -\end{equation*} -$$ - -or consider the system -$$ -\begin{equation*} -\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, -\end{equation*} -$$ - -instead. -

            - - -

            -









            - -

            Conjugate gradient method

            -
            - -

            -One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form -$$ -\begin{equation*} - f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. -\end{equation*} -$$ - -This suggests taking the first basis vector \( \hat{p}_1 \) -to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), -which equals -$$ -\begin{equation*} -\hat{A}\hat{x}_0-\hat{b}, -\end{equation*} -$$ - -and -\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). -The other vectors in the basis will be conjugate to the gradient, -hence the name conjugate gradient method. -

            - - -

            -









            - -

            Conjugate gradient method

            -
            - -

            -Let \( \hat{r}_k \) be the residual at the \( k \)-th step: -$$ -\begin{equation*} -\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k. -\end{equation*} -$$ - -Note that \( \hat{r}_k \) is the negative gradient of \( f \) at -\( \hat{x}=\hat{x}_k \), -so the gradient descent method would be to move in the direction \( \hat{r}_k \). -Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other, -so we take the direction closest to the gradient \( \hat{r}_k \) -under the conjugacy constraint. -This gives the following expression -$$ -\begin{equation*} -\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k. -\end{equation*} -$$ -

            - - -

            -









            - -

            Conjugate gradient method

            -
            - -

            -We can also compute the residual iteratively as -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, - \end{equation*} -$$ - -which equals -$$ -\begin{equation*} -\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k), - \end{equation*} -$$ - -or -$$ -\begin{equation*} -(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k, - \end{equation*} -$$ - -which gives - -$$ -\begin{equation*} -\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k}, - \end{equation*} -$$ -

            - - -

            -









            - -

            Simple implementation of the Conjugate gradient algorithm

            -
            - -

            -

            - - -

              Vector ConjugateGradient(Matrix A, Vector b, Vector x0){
            -  int dim = x0.Dimension();
            -  const double tolerance = 1.0e-14;
            -  Vector x(dim),r(dim),v(dim),z(dim);
            -  double c,t,d;
            -
            -  x = x0;
            -  r = b - A*x;
            -  v = r;
            -  c = dot(r,r);
            -  int i = 0; IterMax = dim;
            -  while(i <= IterMax){
            -    z = A*v;
            -    t = c/dot(v,z);
            -    x = x + t*v;
            -    r = r - t*z;
            -    d = dot(r,r);
            -    if(sqrt(d) < tolerance)
            -      break;
            -    v = r + (d/c)*v;
            -    c = d;  i++;
            -  }
            -  return x;
            -} 
            -
            - -
            - - -

            -









            - -

            Broyden–Fletcher–Goldfarb–Shanno algorithm

            -
            - -

            -The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take. - -

            -The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage. - -

            -The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation -$$ -B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}), -$$ - -

            -where \( B_{k} \) is an approximation to the Hessian matrix, which is -updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \) -is the gradient of the function -evaluated at \( x_k \). -A line search in the direction \( p_k \) is then used to -find the next point \( x_{k+1} \) by minimising -$$ -f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}), -$$ - -over the scalar \( \alpha > 0 \). - - -

            - -

            -

            Revisiting our first homework

            +

            Revisiting our first homework

            We will use linear regression as a case study for the gradient descent @@ -1294,7 +688,7 @@ $$

            -

            Gradient descent example

            +

            Gradient descent example

            Let \( \mathbf{y} = (y_1,\cdots,y_n)^T \), \( \mathbf{\hat{y}} = (\hat{y}_1,\cdots,\hat{y}_n)^T \) and \( \beta = (\beta_0, \beta_1)^T \) @@ -1319,7 +713,7 @@ and we want to find \( \beta \) such that \( C(\beta) \) is minimized.











            -

            The derivative of the cost/loss function

            +

            The derivative of the cost/loss function

            Computing \( \partial C(\beta) / \partial \beta_0 \) and \( \partial C(\beta) / \partial \beta_1 \) we can show that the gradient can be written as @@ -1334,7 +728,7 @@ where \( X \) is the design matrix defined above.











            -

            The Hessian matrix

            +

            The Hessian matrix

            The Hessian matrix of \( C(\beta) \) is given by $$ \hat{H} \equiv \begin{bmatrix} @@ -1348,7 +742,7 @@ This result implies that \( C(\beta) \) is a convex function since the matrix \(











            -

            Simple program

            +

            Simple program

            We can now write a program that minimizes \( C(\beta) \) using the gradient descent method with a constant learning rate \( \gamma \) according to @@ -1389,7 +783,7 @@ beta_NE = np.









            -

            Gradient Descent Example

            +

            Gradient Descent Example

            Another simple example is here @@ -1438,7 +832,7 @@ plt.show()











            -

            And a corresponding example using scikit-learn

            +

            And a corresponding example using scikit-learn

            @@ -1462,7 +856,7 @@ sgdreg.fit(x,y.

            -

            Gradient descent and Ridge

            +

            Gradient descent and Ridge

            We have also discussed Ridge regression where the loss function contains a regularized given by the \( L_2 \) norm of \( \beta \), @@ -1519,393 +913,7 @@ beta_ridge = np











            -

            Automatic differentiation

            -Python has tools for so-called automatic differentiation. -Consider the following example -$$ -f(x) = \sin\left(2\pi x + x^2\right) -$$ - -which has the following derivative -$$ -f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) -$$ - -Using autograd we have - -

            - - -

            import autograd.numpy as np
            -
            -# To do elementwise differentiation:
            -from autograd import elementwise_grad as egrad 
            -
            -# To plot:
            -import matplotlib.pyplot as plt 
            -
            -
            -def f(x):
            -    return np.sin(2*np.pi*x + x**2)
            -
            -def f_grad_analytic(x):
            -    return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
            -
            -# Do the comparison:
            -x = np.linspace(0,1,1000)
            -
            -f_grad = egrad(f)
            -
            -computed = f_grad(x)
            -analytic = f_grad_analytic(x)
            -
            -plt.title('Derivative computed from Autograd compared with the analytical derivative')
            -plt.plot(x,computed,label='autograd')
            -plt.plot(x,analytic,label='analytic')
            -
            -plt.xlabel('x')
            -plt.ylabel('y')
            -plt.legend()
            -
            -plt.show()
            -
            -print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
            -
            -

            - - -

            Using autograd

            - -

            -Here we -experiment with what kind of functions Autograd is capable -of finding the gradient of. The following Python functions are just -meant to illustrate what Autograd can do, but please feel free to -experiment with other, possibly more complicated, functions as well. - -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -
            -def f1(x):
            -    return x**3 + 1
            -
            -f1_grad = grad(f1)
            -
            -# Remember to send in float as argument to the computed gradient from Autograd!
            -a = 1.0
            -
            -# See the evaluated gradient at a using autograd:
            -print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
            -
            -# Compare with the analytical derivative, that is f1'(x) = 3*x**2 
            -grad_analytical = 3*a**2
            -print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
            -
            -

            -









            - -

            Autograd with more complicated functions

            - -

            -To differentiate with respect to two (or more) arguments of a Python -function, Autograd need to know at which variable the function if -being differentiated with respect to. - -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -def f2(x1,x2):
            -    return 3*x1**3 + x2*(x1 - 5) + 1
            -
            -# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
            -f2_grad_x1 = grad(f2,0)
            -
            -# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
            -f2_grad_x2 = grad(f2,1)
            -
            -x1 = 1.0
            -x2 = 3.0 
            -
            -print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
            -print("-"*30)
            -
            -# Compare with the analytical derivatives:
            -
            -# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
            -f2_grad_x1_analytical = 9*x1**2 + x2
            -
            -# Derivative of f2 w.r.t x2 is: x1 - 5:
            -f2_grad_x2_analytical = x1 - 5
            -
            -# See the evaluated derivations:
            -print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
            -print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
            -
            -print()
            -
            -print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
            -print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
            -
            -

            -Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable. - -

            -









            - -

            More complicated functions using the elements of their arguments directly

            - -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -def f3(x): # Assumes x is an array of length 5 or higher
            -    return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
            -
            -f3_grad = grad(f3)
            -
            -x = np.linspace(0,4,5)
            -
            -# Print the computed gradient:
            -print("The computed gradient of f3 is: ", f3_grad(x))
            -
            -# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
            -f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
            -
            -# Print the analytical gradient:
            -print("The analytical gradient of f3 is: ", f3_grad_analytical)
            -
            -

            -Note that in this case, when sending an array as input argument, the -output from Autograd is another array. This is the true gradient of -the function, as opposed to the function in the previous example. By -using arrays to represent the variables, the output from Autograd -might be easier to work with, as the output is closer to what one -could expect form a gradient-evaluting function. - -

            - - -

            Functions using mathematical functions from Numpy

            - -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -def f4(x):
            -    return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
            -
            -f4_grad = grad(f4)
            -
            -x = 2.7
            -
            -# Print the computed derivative:
            -print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
            -
            -# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
            -f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
            -
            -# Print the analytical gradient:
            -print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
            -
            -

            -









            - -

            More autograd

            - -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -def f5(x):
            -    if x >= 0:
            -        return x**2
            -    else:
            -        return -3*x + 1
            -
            -f5_grad = grad(f5)
            -
            -x = 2.7
            -
            -# Print the computed derivative:
            -print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
            -
            -

            -









            - -

            And with loops

            - -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -def f6_for(x):
            -    val = 0
            -    for i in range(10):
            -        val = val + x**i
            -    return val
            -
            -def f6_while(x):
            -    val = 0
            -    i = 0
            -    while i < 10:
            -        val = val + x**i
            -        i = i + 1
            -    return val
            -
            -f6_for_grad = grad(f6_for)
            -f6_while_grad = grad(f6_while)
            -
            -x = 0.5
            -
            -# Print the computed derivaties of f6_for and f6_while
            -print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
            -print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
            -
            -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
            -# The analytical derivative is: sum(i*x**(i-1)) 
            -f6_grad_analytical = 0
            -for i in range(10):
            -    f6_grad_analytical += i*x**(i-1)
            -
            -print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
            -
            -

            -









            - -

            Using recursion

            -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -
            -def f7(n): # Assume that n is an integer
            -    if n == 1 or n == 0:
            -        return 1
            -    else:
            -        return n*f7(n-1)
            -
            -f7_grad = grad(f7)
            -
            -n = 2.0
            -
            -print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
            -
            -# The function f7 is an implementation of the factorial of n.
            -# By using the product rule, one can find that the derivative is:
            -
            -f7_grad_analytical = 0
            -for i in range(int(n)-1):
            -    tmp = 1
            -    for k in range(int(n)-1):
            -        if k != i:
            -            tmp *= (n - k)
            -    f7_grad_analytical += tmp
            -
            -print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
            -
            -

            -Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input. - -

            -









            - -

            Unsupported functions

            -Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd. - -

            -Assigning a value to the variable being differentiated with respect to -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -def f8(x): # Assume x is an array
            -    x[2] = 3
            -    return x*2
            -
            -f8_grad = grad(f8)
            -
            -x = 8.4
            -
            -print("The derivative of f8 is:",f8_grad(x))
            -
            -

            -Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible. - -

            -









            - -

            The syntax a.dot(b) when finding the dot product

            -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -def f9(a): # Assume a is an array with 2 elements
            -    b = np.array([1.0,2.0])
            -    return a.dot(b)
            -
            -f9_grad = grad(f9)
            -
            -x = np.array([1.0,0.0])
            -
            -print("The derivative of f9 is:",f9_grad(x))
            -
            -

            -Here we are told that the 'dot' function does not belong to Autograd's -version of a Numpy array. To overcome this, an alternative syntax -which also computed the dot product can be used: - -

            - - -

            import autograd.numpy as np
            -from autograd import grad
            -def f9_alternative(x): # Assume a is an array with 2 elements
            -    b = np.array([1.0,2.0])
            -    return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
            -
            -f9_alternative_grad = grad(f9_alternative)
            -
            -x = np.array([3.0,0.0])
            -
            -print("The gradient of f9 is:",f9_alternative_grad(x))
            -
            -# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
            -# w.r.t x is (b_1, b_2).
            -
            -

            -









            - -

            Recommended to avoid

            -The documentation recommends to avoid inplace operations such as -

            - - -

            a += b
            -a -= b
            -a*= b
            -a /=b
            -
            -

            -









            - -

            Stochastic Gradient Descent

            +

            Stochastic Gradient Descent

            Stochastic gradient descent (SGD) and variants thereof address some of @@ -1923,7 +931,7 @@ $$











            -

            Computation of gradients

            +

            Computation of gradients

            This in turn means that the gradient can be @@ -1943,7 +951,7 @@ minibatches. We denote these minibatches by \( B_k \) where











            -

            SGD example

            +

            SGD example

            As an example, suppose we have \( 10 \) data points \( (\mathbf{x}_1,\cdots, \mathbf{x}_{10}) \) and we choose to have \( M=5 \) minibathces, then each minibatch contains two data points. In particular we have @@ -1967,7 +975,7 @@ $$











            -

            The gradient step

            +

            The gradient step

            Thus a gradient descent step now looks like @@ -1986,7 +994,7 @@ the number of minibatches, as exemplified in the code below.











            -

            Simple example code

            +

            Simple example code

            @@ -2018,7 +1026,7 @@ all \( n \) datapoints.











            -

            When do we stop?

            +

            When do we stop?

            A natural question is when do we stop the search for a new minimum? @@ -2035,7 +1043,7 @@ gave the lowest value.











            -

            Slightly different approach

            +

            Slightly different approach

            Another approach is to let the step length \( \gamma_j \) depend on the @@ -2083,7 +1091,7 @@ j = 0











            -

            Program for stochastic gradient

            +

            Program for stochastic gradient

            @@ -2162,7 +1170,7 @@ plt.show()











            -

            Using gradient descent methods, limitations

            +

            Using gradient descent methods, limitations

            • Gradient descent (GD) finds local minima of our function. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our energy function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.
            • @@ -2175,7 +1183,7 @@ plt.show()









              -

              Momentum based GD

              +

              Momentum based GD

              The stochastic gradient descent (SGD) is almost always used with a momentum or inertia term that serves as a memory of the direction we are moving in parameter space. This is typically @@ -2198,7 +1206,7 @@ where we have defined \( \Delta \boldsymbol{\theta}_{t}= \boldsymbol{\theta}_t-\











              -

              More on momentum based approaches

              +

              More on momentum based approaches

              Let us try to get more intuition from these equations. It is helpful to consider a simple physical analogy with a particle of mass \( m \) moving in a viscous medium with drag coefficient \( \mu \) and potential @@ -2220,7 +1228,7 @@ $$











              -

              Momentum parameter

              +

              Momentum parameter

              Notice that this equation is identical to previous one if we identify the position of the particle, \( \mathbf{w} \), with the parameters \( \boldsymbol{\theta} \). This allows us to identify the momentum parameter and learning rate with the mass of the particle and the viscous drag as: $$ @@ -2250,7 +1258,7 @@ One of the major advantages of NAG is that it allows for the use of a larger lea











              -

              Second moment of the gradient

              +

              Second moment of the gradient

              In stochastic gradient descent, with and without momentum, we still @@ -2275,7 +1283,7 @@ Recently, a number of methods have been introduced that accomplish this by track











              -

              RMS prop

              +

              RMS prop

              In RMS prop, in addition to keeping a running average of the first moment of the gradient, we also keep track of the second moment denoted by \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}_t^2] \). The update rule for RMS prop is given by @@ -2293,7 +1301,7 @@ where \( \beta \) controls the averaging time of the second moment and is typica











              -

              ADAM optimizer

              +

              ADAM optimizer

              A related algorithm is the ADAM optimizer. In ADAM, we keep a running average of both the first and second moment of the gradient and use this information to adaptively change the learning rate for different parameters. In addition to keeping a running average of the first and second moments of the gradient (i.e. \( \mathbf{m}_t=\mathbb{E}[\mathbf{g}_t] \) and \( \mathbf{s}_t=\mathbb{E}[\mathbf{g}^2_t] \), respectively), ADAM performs an additional bias correction to account for the fact that we are estimating the first two moments of the gradient using a running average (denoted by the hats in the update rule below). The update rule for ADAM is given by (where multiplication and division are once again understood to be element-wise operations below) @@ -2321,7 +1329,7 @@ $$











              -

              Practical tips

              +

              Practical tips

              • Randomize the data when making mini-batches. It is always important to randomly shuffle the data when forming mini-batches. Otherwise, the gradient descent method can fit spurious correlations resulting from the order in which data is presented.
              • @@ -2332,11 +1340,1012 @@ $$ Geron's text, see chapter 11, has several interesting discussions. +

                +









                + +

                Automatic differentiation

                +Python has tools for so-called automatic differentiation. +Consider the following example +$$ +f(x) = \sin\left(2\pi x + x^2\right) +$$ + +which has the following derivative +$$ +f'(x) = \cos\left(2\pi x + x^2\right)\left(2\pi + 2x\right) +$$ + +Using autograd we have + +

                + + +

                import autograd.numpy as np
                +
                +# To do elementwise differentiation:
                +from autograd import elementwise_grad as egrad 
                +
                +# To plot:
                +import matplotlib.pyplot as plt 
                +
                +
                +def f(x):
                +    return np.sin(2*np.pi*x + x**2)
                +
                +def f_grad_analytic(x):
                +    return np.cos(2*np.pi*x + x**2)*(2*np.pi + 2*x)
                +
                +# Do the comparison:
                +x = np.linspace(0,1,1000)
                +
                +f_grad = egrad(f)
                +
                +computed = f_grad(x)
                +analytic = f_grad_analytic(x)
                +
                +plt.title('Derivative computed from Autograd compared with the analytical derivative')
                +plt.plot(x,computed,label='autograd')
                +plt.plot(x,analytic,label='analytic')
                +
                +plt.xlabel('x')
                +plt.ylabel('y')
                +plt.legend()
                +
                +plt.show()
                +
                +print("The max absolute difference is: %g"%(np.max(np.abs(computed - analytic))))
                +
                +

                + + +

                Using autograd

                + +

                +Here we +experiment with what kind of functions Autograd is capable +of finding the gradient of. The following Python functions are just +meant to illustrate what Autograd can do, but please feel free to +experiment with other, possibly more complicated, functions as well. + +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +
                +def f1(x):
                +    return x**3 + 1
                +
                +f1_grad = grad(f1)
                +
                +# Remember to send in float as argument to the computed gradient from Autograd!
                +a = 1.0
                +
                +# See the evaluated gradient at a using autograd:
                +print("The gradient of f1 evaluated at a = %g using autograd is: %g"%(a,f1_grad(a)))
                +
                +# Compare with the analytical derivative, that is f1'(x) = 3*x**2 
                +grad_analytical = 3*a**2
                +print("The gradient of f1 evaluated at a = %g by finding the analytic expression is: %g"%(a,grad_analytical))
                +
                +

                +









                + +

                Autograd with more complicated functions

                + +

                +To differentiate with respect to two (or more) arguments of a Python +function, Autograd need to know at which variable the function if +being differentiated with respect to. + +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +def f2(x1,x2):
                +    return 3*x1**3 + x2*(x1 - 5) + 1
                +
                +# By sending the argument 0, Autograd will compute the derivative w.r.t the first variable, in this case x1
                +f2_grad_x1 = grad(f2,0)
                +
                +# ... and differentiate w.r.t x2 by sending 1 as an additional arugment to grad
                +f2_grad_x2 = grad(f2,1)
                +
                +x1 = 1.0
                +x2 = 3.0 
                +
                +print("Evaluating at x1 = %g, x2 = %g"%(x1,x2))
                +print("-"*30)
                +
                +# Compare with the analytical derivatives:
                +
                +# Derivative of f2 w.r.t x1 is: 9*x1**2 + x2:
                +f2_grad_x1_analytical = 9*x1**2 + x2
                +
                +# Derivative of f2 w.r.t x2 is: x1 - 5:
                +f2_grad_x2_analytical = x1 - 5
                +
                +# See the evaluated derivations:
                +print("The derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
                +print("The analytical derivative of f2 w.r.t x1: %g"%( f2_grad_x1(x1,x2) ))
                +
                +print()
                +
                +print("The derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
                +print("The analytical derivative of f2 w.r.t x2: %g"%( f2_grad_x2(x1,x2) ))
                +
                +

                +Note that the grad function will not produce the true gradient of the function. The true gradient of a function with two or more variables will produce a vector, where each element is the function differentiated w.r.t a variable. + +

                +









                + +

                More complicated functions using the elements of their arguments directly

                + +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +def f3(x): # Assumes x is an array of length 5 or higher
                +    return 2*x[0] + 3*x[1] + 5*x[2] + 7*x[3] + 11*x[4]**2
                +
                +f3_grad = grad(f3)
                +
                +x = np.linspace(0,4,5)
                +
                +# Print the computed gradient:
                +print("The computed gradient of f3 is: ", f3_grad(x))
                +
                +# The analytical gradient is: (2, 3, 5, 7, 22*x[4])
                +f3_grad_analytical = np.array([2, 3, 5, 7, 22*x[4]])
                +
                +# Print the analytical gradient:
                +print("The analytical gradient of f3 is: ", f3_grad_analytical)
                +
                +

                +Note that in this case, when sending an array as input argument, the +output from Autograd is another array. This is the true gradient of +the function, as opposed to the function in the previous example. By +using arrays to represent the variables, the output from Autograd +might be easier to work with, as the output is closer to what one +could expect form a gradient-evaluting function. + +

                + + +

                Functions using mathematical functions from Numpy

                + +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +def f4(x):
                +    return np.sqrt(1+x**2) + np.exp(x) + np.sin(2*np.pi*x)
                +
                +f4_grad = grad(f4)
                +
                +x = 2.7
                +
                +# Print the computed derivative:
                +print("The computed derivative of f4 at x = %g is: %g"%(x,f4_grad(x)))
                +
                +# The analytical derivative is: x/sqrt(1 + x**2) + exp(x) + cos(2*pi*x)*2*pi
                +f4_grad_analytical = x/np.sqrt(1 + x**2) + np.exp(x) + np.cos(2*np.pi*x)*2*np.pi
                +
                +# Print the analytical gradient:
                +print("The analytical gradient of f4 at x = %g is: %g"%(x,f4_grad_analytical))
                +
                +

                +









                + +

                More autograd

                + +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +def f5(x):
                +    if x >= 0:
                +        return x**2
                +    else:
                +        return -3*x + 1
                +
                +f5_grad = grad(f5)
                +
                +x = 2.7
                +
                +# Print the computed derivative:
                +print("The computed derivative of f5 at x = %g is: %g"%(x,f5_grad(x)))
                +
                +

                +









                + +

                And with loops

                + +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +def f6_for(x):
                +    val = 0
                +    for i in range(10):
                +        val = val + x**i
                +    return val
                +
                +def f6_while(x):
                +    val = 0
                +    i = 0
                +    while i < 10:
                +        val = val + x**i
                +        i = i + 1
                +    return val
                +
                +f6_for_grad = grad(f6_for)
                +f6_while_grad = grad(f6_while)
                +
                +x = 0.5
                +
                +# Print the computed derivaties of f6_for and f6_while
                +print("The computed derivative of f6_for at x = %g is: %g"%(x,f6_for_grad(x)))
                +print("The computed derivative of f6_while at x = %g is: %g"%(x,f6_while_grad(x)))
                +
                +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +# Both of the functions are implementation of the sum: sum(x**i) for i = 0, ..., 9
                +# The analytical derivative is: sum(i*x**(i-1)) 
                +f6_grad_analytical = 0
                +for i in range(10):
                +    f6_grad_analytical += i*x**(i-1)
                +
                +print("The analytical derivative of f6 at x = %g is: %g"%(x,f6_grad_analytical))
                +
                +

                +









                + +

                Using recursion

                +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +
                +def f7(n): # Assume that n is an integer
                +    if n == 1 or n == 0:
                +        return 1
                +    else:
                +        return n*f7(n-1)
                +
                +f7_grad = grad(f7)
                +
                +n = 2.0
                +
                +print("The computed derivative of f7 at n = %d is: %g"%(n,f7_grad(n)))
                +
                +# The function f7 is an implementation of the factorial of n.
                +# By using the product rule, one can find that the derivative is:
                +
                +f7_grad_analytical = 0
                +for i in range(int(n)-1):
                +    tmp = 1
                +    for k in range(int(n)-1):
                +        if k != i:
                +            tmp *= (n - k)
                +    f7_grad_analytical += tmp
                +
                +print("The analytical derivative of f7 at n = %d is: %g"%(n,f7_grad_analytical))
                +
                +

                +Note that if n is equal to zero or one, Autograd will give an error message. This message appears when the output is independent on input. + +

                +









                + +

                Unsupported functions

                +Autograd supports many features. However, there are some functions that is not supported (yet) by Autograd. + +

                +Assigning a value to the variable being differentiated with respect to +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +def f8(x): # Assume x is an array
                +    x[2] = 3
                +    return x*2
                +
                +f8_grad = grad(f8)
                +
                +x = 8.4
                +
                +print("The derivative of f8 is:",f8_grad(x))
                +
                +

                +Here, Autograd tells us that an 'ArrayBox' does not support item assignment. The item assignment is done when the program tries to assign x[2] to the value 3. However, Autograd has implemented the computation of the derivative such that this assignment is not possible. + +

                +









                + +

                The syntax a.dot(b) when finding the dot product

                +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +def f9(a): # Assume a is an array with 2 elements
                +    b = np.array([1.0,2.0])
                +    return a.dot(b)
                +
                +f9_grad = grad(f9)
                +
                +x = np.array([1.0,0.0])
                +
                +print("The derivative of f9 is:",f9_grad(x))
                +
                +

                +Here we are told that the 'dot' function does not belong to Autograd's +version of a Numpy array. To overcome this, an alternative syntax +which also computed the dot product can be used: + +

                + + +

                import autograd.numpy as np
                +from autograd import grad
                +def f9_alternative(x): # Assume a is an array with 2 elements
                +    b = np.array([1.0,2.0])
                +    return np.dot(x,b) # The same as x_1*b_1 + x_2*b_2
                +
                +f9_alternative_grad = grad(f9_alternative)
                +
                +x = np.array([3.0,0.0])
                +
                +print("The gradient of f9 is:",f9_alternative_grad(x))
                +
                +# The analytical gradient of the dot product of vectors x and b with two elements (x_1,x_2) and (b_1, b_2) respectively
                +# w.r.t x is (b_1, b_2).
                +
                +

                +









                + +

                Recommended to avoid

                +The documentation recommends to avoid inplace operations such as +

                + + +

                a += b
                +a -= b
                +a*= b
                +a /=b
                +
                +

                +









                + +

                Standard steepest descent

                + +

                +Before we proceed, we would like to discuss the approach called the +standard Steepest descent, which again leads to us having to be able +to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG). + +

                +The success of the CG method +for finding solutions of non-linear problems is based on the theory +of conjugate gradients for linear systems of equations. It belongs to +the class of iterative methods for solving problems from linear +algebra of the type +$$ +\begin{equation*} +\hat{A}\hat{x} = \hat{b}. +\end{equation*} +$$ + +

                +In the iterative process we end up with a problem like + +$$ +\begin{equation*} + \hat{r}= \hat{b}-\hat{A}\hat{x}, +\end{equation*} +$$ + +where \( \hat{r} \) is the so-called residual or error in the iterative process. + +

                +When we have found the exact solution, \( \hat{r}=0 \). + +

                +









                + +

                Gradient method

                + +

                +The residual is zero when we reach the minimum of the quadratic equation +$$ +\begin{equation*} + P(\hat{x})=\frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T\hat{b}, +\end{equation*} +$$ + +

                +with the constraint that the matrix \( \hat{A} \) is positive definite and +symmetric. This defines also the Hessian and we want it to be positive definite. + +

                +









                + +

                Steepest descent method

                + +

                +We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). +We can assume without loss of generality that +$$ +\begin{equation*} +\hat{x}_0=0, +\end{equation*} +$$ + +or consider the system +$$ +\begin{equation*} +\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, +\end{equation*} +$$ + +instead. + +

                +









                + +

                Steepest descent method

                +
                + +

                +One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form +$$ +\begin{equation*} + f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. +\end{equation*} +$$ + +This suggests taking the first basis vector \( \hat{r}_1 \) (see below for definition) +to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), +which equals +$$ +\begin{equation*} +\hat{A}\hat{x}_0-\hat{b}, +\end{equation*} +$$ + +and +\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). + + +

                + + +

                +









                + +

                Final expressions

                +
                + +

                +We can compute the residual iteratively as +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, + \end{equation*} +$$ + +which equals +$$ +\begin{equation*} +\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{r}_k), + \end{equation*} +$$ + +or +$$ +\begin{equation*} +(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{r}_k, + \end{equation*} +$$ + +which gives + +$$ +\alpha_k = \frac{\hat{r}_k^T\hat{r}_k}{\hat{r}_k^T\hat{A}\hat{r}_k} +$$ + +leading to the iterative scheme +$$ +\begin{equation*} +\hat{x}_{k+1}=\hat{x}_k-\alpha_k\hat{r}_{k}, + \end{equation*} +$$ +

                + + +

                +









                + +

                Code examples for steepest descent

                + +

                +









                + +

                Simple codes for steepest descent and conjugate gradient using a \( 2\times 2 \) matrix, in c++, Python code to come

                +
                + +

                +

                + + +

                #include <cmath>
                +#include <iostream>
                +#include <fstream>
                +#include <iomanip>
                +#include "vectormatrixclass.h"
                +using namespace  std;
                +//   Main function begins here
                +int main(int  argc, char * argv[]){
                +  int dim = 2;
                +  Vector x(dim),xsd(dim), b(dim),x0(dim);
                +  Matrix A(dim,dim);
                +
                +  // Set our initial guess
                +  x0(0) = x0(1) = 0;
                +  // Set the matrix
                +  A(0,0) =  3;    A(1,0) =  2;   A(0,1) =  2;   A(1,1) =  6;
                +  b(0) = 2; b(1) = -8;
                +  cout << "The Matrix A that we are using: " << endl;
                +  A.Print();
                +  cout << endl;
                +  xsd = SteepestDescent(A,b,x0);
                +  cout << "The approximate solution using Steepest Descent is: " << endl;
                +  xsd.Print();
                +  cout << endl;
                +}
                +
                + +
                + + +

                +









                + +

                The routine for the steepest descent method

                +
                + +

                +

                + + +

                Vector SteepestDescent(Matrix A, Vector b, Vector x0){
                +  int IterMax, i;
                +  int dim = x0.Dimension();
                +  const double tolerance = 1.0e-14;
                +  Vector x(dim),f(dim),z(dim);
                +  double c,alpha,d;
                +  IterMax = 30;
                +  x = x0;
                +  r = A*x-b;
                +  i = 0;
                +  while (i <= IterMax){
                +    z = A*r;
                +    c = dot(r,r);
                +    alpha = c/dot(r,z);
                +    x = x - alpha*r;
                +    r =  A*x-b;
                +    if(sqrt(dot(r,r)) < tolerance) break;
                +    i++;
                +  }
                +  return x;
                +}
                +
                + +
                + + +

                +









                + +

                Steepest descent example

                + +

                + + +

                import numpy as np
                +import numpy.linalg as la
                +
                +import scipy.optimize as sopt
                +
                +import matplotlib.pyplot as pt
                +from mpl_toolkits.mplot3d import axes3d
                +
                +def f(x):
                +    return 0.5*x[0]**2 + 2.5*x[1]**2
                +
                +def df(x):
                +    return np.array([x[0], 5*x[1]])
                +
                +fig = pt.figure()
                +ax = fig.gca(projection="3d")
                +
                +xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]
                +fmesh = f(np.array([xmesh, ymesh]))
                +ax.plot_surface(xmesh, ymesh, fmesh)
                +
                +

                +And then as countor plot +

                + + +

                pt.axis("equal")
                +pt.contour(xmesh, ymesh, fmesh)
                +guesses = [np.array([2, 2./5])]
                +
                +

                +Find guesses +

                + + +

                x = guesses[-1]
                +s = -df(x)
                +
                +

                +Run it! +

                + + +

                def f1d(alpha):
                +    return f(x + alpha*s)
                +
                +alpha_opt = sopt.golden(f1d)
                +next_guess = x + alpha_opt * s
                +guesses.append(next_guess)
                +print(next_guess)
                +
                +

                +What happened? +

                + + +

                pt.axis("equal")
                +pt.contour(xmesh, ymesh, fmesh, 50)
                +it_array = np.array(guesses)
                +pt.plot(it_array.T[0], it_array.T[1], "x-")
                +
                +

                +









                + +

                Conjugate gradient method

                +
                + +

                +In the CG method we define so-called conjugate directions and two vectors +\( \hat{s} \) and \( \hat{t} \) +are said to be +conjugate if +$$ +\begin{equation*} +\hat{s}^T\hat{A}\hat{t}= 0. +\end{equation*} +$$ + +The philosophy of the CG method is to perform searches in various conjugate directions +of our vectors \( \hat{x}_i \) obeying the above criterion, namely +$$ +\begin{equation*} +\hat{x}_i^T\hat{A}\hat{x}_j= 0. +\end{equation*} +$$ + +Two vectors are conjugate if they are orthogonal with respect to +this inner product. Being conjugate is a symmetric relation: if \( \hat{s} \) is conjugate to \( \hat{t} \), then \( \hat{t} \) is conjugate to \( \hat{s} \). +

                + + +

                +









                + +

                Conjugate gradient method

                +
                + +

                +An example is given by the eigenvectors of the matrix +$$ +\begin{equation*} +\hat{v}_i^T\hat{A}\hat{v}_j= \lambda\hat{v}_i^T\hat{v}_j, +\end{equation*} +$$ + +which is zero unless \( i=j \). +

                + + +

                +









                + +

                Conjugate gradient method

                +
                + +

                +Assume now that we have a symmetric positive-definite matrix \( \hat{A} \) of size +\( n\times n \). At each iteration \( i+1 \) we obtain the conjugate direction of a vector +$$ +\begin{equation*} +\hat{x}_{i+1}=\hat{x}_{i}+\alpha_i\hat{p}_{i}. +\end{equation*} +$$ + +We assume that \( \hat{p}_{i} \) is a sequence of \( n \) mutually conjugate directions. +Then the \( \hat{p}_{i} \) form a basis of \( R^n \) and we can expand the solution +$ \hat{A}\hat{x} = \hat{b}$ in this basis, namely + +$$ +\begin{equation*} + \hat{x} = \sum^{n}_{i=1} \alpha_i \hat{p}_i. +\end{equation*} +$$ +

                + + +

                +









                + +

                Conjugate gradient method

                +
                + +

                +The coefficients are given by +$$ +\begin{equation*} + \mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. +\end{equation*} +$$ + +Multiplying with \( \hat{p}_k^T \) from the left gives + +$$ +\begin{equation*} + \hat{p}_k^T \hat{A}\hat{x} = \sum^{n}_{i=1} \alpha_i\hat{p}_k^T \hat{A}\hat{p}_i= \hat{p}_k^T \hat{b}, +\end{equation*} +$$ + +and we can define the coefficients \( \alpha_k \) as + +$$ +\begin{equation*} + \alpha_k = \frac{\hat{p}_k^T \hat{b}}{\hat{p}_k^T \hat{A} \hat{p}_k} +\end{equation*} +$$ +

                + + +

                +









                + +

                Conjugate gradient method and iterations

                +
                + +

                + +

                +If we choose the conjugate vectors \( \hat{p}_k \) carefully, +then we may not need all of them to obtain a good approximation to the solution +\( \hat{x} \). +We want to regard the conjugate gradient method as an iterative method. +This will us to solve systems where \( n \) is so large that the direct +method would take too much time. + +

                +We denote the initial guess for \( \hat{x} \) as \( \hat{x}_0 \). +We can assume without loss of generality that +$$ +\begin{equation*} +\hat{x}_0=0, +\end{equation*} +$$ + +or consider the system +$$ +\begin{equation*} +\hat{A}\hat{z} = \hat{b}-\hat{A}\hat{x}_0, +\end{equation*} +$$ + +instead. +

                + + +

                +









                + +

                Conjugate gradient method

                +
                + +

                +One can show that the solution \( \hat{x} \) is also the unique minimizer of the quadratic form +$$ +\begin{equation*} + f(\hat{x}) = \frac{1}{2}\hat{x}^T\hat{A}\hat{x} - \hat{x}^T \hat{x} , \quad \hat{x}\in\mathbf{R}^n. +\end{equation*} +$$ + +This suggests taking the first basis vector \( \hat{p}_1 \) +to be the gradient of \( f \) at \( \hat{x}=\hat{x}_0 \), +which equals +$$ +\begin{equation*} +\hat{A}\hat{x}_0-\hat{b}, +\end{equation*} +$$ + +and +\( \hat{x}_0=0 \) it is equal \( -\hat{b} \). +The other vectors in the basis will be conjugate to the gradient, +hence the name conjugate gradient method. +

                + + +

                +









                + +

                Conjugate gradient method

                +
                + +

                +Let \( \hat{r}_k \) be the residual at the \( k \)-th step: +$$ +\begin{equation*} +\hat{r}_k=\hat{b}-\hat{A}\hat{x}_k. +\end{equation*} +$$ + +Note that \( \hat{r}_k \) is the negative gradient of \( f \) at +\( \hat{x}=\hat{x}_k \), +so the gradient descent method would be to move in the direction \( \hat{r}_k \). +Here, we insist that the directions \( \hat{p}_k \) are conjugate to each other, +so we take the direction closest to the gradient \( \hat{r}_k \) +under the conjugacy constraint. +This gives the following expression +$$ +\begin{equation*} +\hat{p}_{k+1}=\hat{r}_k-\frac{\hat{p}_k^T \hat{A}\hat{r}_k}{\hat{p}_k^T\hat{A}\hat{p}_k} \hat{p}_k. +\end{equation*} +$$ +

                + + +

                +









                + +

                Conjugate gradient method

                +
                + +

                +We can also compute the residual iteratively as +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{b}-\hat{A}\hat{x}_{k+1}, + \end{equation*} +$$ + +which equals +$$ +\begin{equation*} +\hat{b}-\hat{A}(\hat{x}_k+\alpha_k\hat{p}_k), + \end{equation*} +$$ + +or +$$ +\begin{equation*} +(\hat{b}-\hat{A}\hat{x}_k)-\alpha_k\hat{A}\hat{p}_k, + \end{equation*} +$$ + +which gives + +$$ +\begin{equation*} +\hat{r}_{k+1}=\hat{r}_k-\hat{A}\hat{p}_{k}, + \end{equation*} +$$ +

                + + +

                +









                + +

                Simple implementation of the Conjugate gradient algorithm

                +
                + +

                +

                + + +

                  Vector ConjugateGradient(Matrix A, Vector b, Vector x0){
                +  int dim = x0.Dimension();
                +  const double tolerance = 1.0e-14;
                +  Vector x(dim),r(dim),v(dim),z(dim);
                +  double c,t,d;
                +
                +  x = x0;
                +  r = b - A*x;
                +  v = r;
                +  c = dot(r,r);
                +  int i = 0; IterMax = dim;
                +  while(i <= IterMax){
                +    z = A*v;
                +    t = c/dot(v,z);
                +    x = x + t*v;
                +    r = r - t*z;
                +    d = dot(r,r);
                +    if(sqrt(d) < tolerance)
                +      break;
                +    v = r + (d/c)*v;
                +    c = d;  i++;
                +  }
                +  return x;
                +} 
                +
                + +
                + + +

                +









                + +

                Broyden–Fletcher–Goldfarb–Shanno algorithm

                +
                + +

                +The optimization problem is to minimize \( f(\mathbf {x} ) \) where \( \mathbf {x} \) is a vector in \( R^{n} \), and \( f \) is a differentiable scalar function. There are no constraints on the values that \( \mathbf {x} \) can take. + +

                +The algorithm begins at an initial estimate for the optimal value \( \mathbf {x}_{0} \) and proceeds iteratively to get a better estimate at each stage. + +

                +The search direction \( p_k \) at stage \( k \) is given by the solution of the analogue of the Newton equation +$$ +B_{k}\mathbf {p} _{k}=-\nabla f(\mathbf {x}_{k}), +$$ + +

                +where \( B_{k} \) is an approximation to the Hessian matrix, which is +updated iteratively at each stage, and \( \nabla f(\mathbf {x} _{k}) \) +is the gradient of the function +evaluated at \( x_k \). +A line search in the direction \( p_k \) is then used to +find the next point \( x_{k+1} \) by minimising +$$ +f(\mathbf {x}_{k}+\alpha \mathbf {p}_{k}), +$$ + +over the scalar \( \alpha > 0 \). + + +

                + + +

                +

                - © 1999-2018, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license + © 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license
                diff --git a/doc/pub/Splines/html/reveal.js/.gitignore b/doc/pub/Splines/html/reveal.js/.gitignore index e7b4f216a..a5df3133d 100644 --- a/doc/pub/Splines/html/reveal.js/.gitignore +++ b/doc/pub/Splines/html/reveal.js/.gitignore @@ -1,8 +1,3 @@ -.idea/ -*.iml -*.iws -*.eml -out/ .DS_Store .svn log/*.log @@ -10,4 +5,4 @@ tmp/** node_modules/ .sass-cache css/reveal.min.css -js/reveal.min.js \ No newline at end of file +js/reveal.min.js diff --git a/doc/pub/Splines/html/reveal.js/.travis.yml b/doc/pub/Splines/html/reveal.js/.travis.yml index ec3b27d5d..165d9ae9f 100644 --- a/doc/pub/Splines/html/reveal.js/.travis.yml +++ b/doc/pub/Splines/html/reveal.js/.travis.yml @@ -1,7 +1,5 @@ language: node_js node_js: - - 4 + - 0.10 before_script: - - npm install -g grunt-cli -after_script: - - grunt retire + - npm install -g grunt-cli \ No newline at end of file diff --git a/doc/pub/Splines/html/reveal.js/LICENSE b/doc/pub/Splines/html/reveal.js/LICENSE index c3e6e5fd6..09623076f 100644 --- a/doc/pub/Splines/html/reveal.js/LICENSE +++ b/doc/pub/Splines/html/reveal.js/LICENSE @@ -1,4 +1,4 @@ -Copyright (C) 2017 Hakim El Hattab, http://hakim.se, and reveal.js contributors +Copyright (C) 2015 Hakim El Hattab, http://hakim.se Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal diff --git a/doc/pub/Splines/html/reveal.js/README.md b/doc/pub/Splines/html/reveal.js/README.md index f2ab6ca88..573b19597 100644 --- a/doc/pub/Splines/html/reveal.js/README.md +++ b/doc/pub/Splines/html/reveal.js/README.md @@ -1,58 +1,12 @@ -# reveal.js [![Build Status](https://travis-ci.org/hakimel/reveal.js.svg?branch=master)](https://travis-ci.org/hakimel/reveal.js) Slides +# reveal.js [![Build Status](https://travis-ci.org/hakimel/reveal.js.svg?branch=master)](https://travis-ci.org/hakimel/reveal.js) -A framework for easily creating beautiful presentations using HTML. [Check out the live demo](http://revealjs.com/). +A framework for easily creating beautiful presentations using HTML. [Check out the live demo](http://lab.hakim.se/reveal-js/). -reveal.js comes with a broad range of features including [nested slides](https://github.com/hakimel/reveal.js#markup), [Markdown contents](https://github.com/hakimel/reveal.js#markdown), [PDF export](https://github.com/hakimel/reveal.js#pdf-export), [speaker notes](https://github.com/hakimel/reveal.js#speaker-notes) and a [JavaScript API](https://github.com/hakimel/reveal.js#api). There's also a fully featured visual editor and platform for sharing reveal.js presentations at [slides.com](https://slides.com?ref=github). +reveal.js comes with a broad range of features including [nested slides](https://github.com/hakimel/reveal.js#markup), [Markdown contents](https://github.com/hakimel/reveal.js#markdown), [PDF export](https://github.com/hakimel/reveal.js#pdf-export), [speaker notes](https://github.com/hakimel/reveal.js#speaker-notes) and a [JavaScript API](https://github.com/hakimel/reveal.js#api). It's best viewed in a modern browser but [fallbacks](https://github.com/hakimel/reveal.js/wiki/Browser-Support) are available to make sure your presentation can still be viewed elsewhere. -## Table of contents -- [Online Editor](#online-editor) -- [Instructions](#instructions) - - [Markup](#markup) - - [Markdown](#markdown) - - [Element Attributes](#element-attributes) - - [Slide Attributes](#slide-attributes) -- [Configuration](#configuration) -- [Presentation Size](#presentation-size) -- [Dependencies](#dependencies) -- [Ready Event](#ready-event) -- [Auto-sliding](#auto-sliding) -- [Keyboard Bindings](#keyboard-bindings) -- [Touch Navigation](#touch-navigation) -- [Lazy Loading](#lazy-loading) -- [API](#api) - - [Slide Changed Event](#slide-changed-event) - - [Presentation State](#presentation-state) - - [Slide States](#slide-states) - - [Slide Backgrounds](#slide-backgrounds) - - [Parallax Background](#parallax-background) - - [Slide Transitions](#slide-transitions) - - [Internal links](#internal-links) - - [Fragments](#fragments) - - [Fragment events](#fragment-events) - - [Code syntax highlighting](#code-syntax-highlighting) - - [Slide number](#slide-number) - - [Overview mode](#overview-mode) - - [Fullscreen mode](#fullscreen-mode) - - [Embedded media](#embedded-media) - - [Stretching elements](#stretching-elements) - - [postMessage API](#postmessage-api) -- [PDF Export](#pdf-export) -- [Theming](#theming) -- [Speaker Notes](#speaker-notes) - - [Share and Print Speaker Notes](#share-and-print-speaker-notes) - - [Server Side Speaker Notes](#server-side-speaker-notes) -- [Multiplexing](#multiplexing) - - [Master presentation](#master-presentation) - - [Client presentation](#client-presentation) - - [Socket.io server](#socketio-server) -- [MathJax](#mathjax) -- [Installation](#installation) - - [Basic setup](#basic-setup) - - [Full setup](#full-setup) - - [Folder Structure](#folder-structure) -- [License](#license) -#### More reading +#### More reading: +- [Installation](#installation): Step-by-step instructions for getting reveal.js running on your computer. - [Changelog](https://github.com/hakimel/reveal.js/releases): Up-to-date version history. - [Examples](https://github.com/hakimel/reveal.js/wiki/Example-Presentations): Presentations created with reveal.js, add your own! - [Browser Support](https://github.com/hakimel/reveal.js/wiki/Browser-Support): Explanation of browser support and fallbacks. @@ -60,36 +14,14 @@ reveal.js comes with a broad range of features including [nested slides](https:/ ## Online Editor -Presentations are written using HTML or Markdown but there's also an online editor for those of you who prefer a graphical interface. Give it a try at [https://slides.com](https://slides.com?ref=github). +Presentations are written using HTML or Markdown but there's also an online editor for those of you who prefer a graphical interface. Give it a try at [http://slides.com](http://slides.com). ## Instructions ### Markup -Here's a barebones example of a fully working reveal.js presentation: -```html - - - - - - -
                -
                -
                Slide 1
                -
                Slide 2
                -
                -
                - - - - -``` - -The presentation markup hierarchy needs to be `.reveal > .slides > section` where the `section` represents one slide and can be repeated indefinitely. If you place multiple `section` elements inside of another `section` they will be shown as vertical slides. The first of the vertical slides is the "root" of the others (at the top), and will be included in the horizontal sequence. For example: +Markup hierarchy needs to be ``
                `` where the ``
                `` represents one slide and can be repeated indefinitely. If you place multiple ``
                ``'s inside of another ``
                `` they will be shown as vertical slides. The first of the vertical slides is the "root" of the others (at the top), and it will be included in the horizontal sequence. For example: ```html
                @@ -105,36 +37,32 @@ The presentation markup hierarchy needs to be `.reveal > .slides > section` wher ### Markdown -It's possible to write your slides using Markdown. To enable Markdown, add the `data-markdown` attribute to your `
                ` elements and wrap the contents in a ` +
                ``` #### External Markdown -You can write your content as a separate file and have reveal.js load it at runtime. Note the separator arguments which determine how slides are delimited in the external file: the `data-separator` attribute defines a regular expression for horizontal slides (defaults to `^\r?\n---\r?\n$`, a newline-bounded horizontal rule) and `data-separator-vertical` defines vertical slides (disabled by default). The `data-separator-notes` attribute is a regular expression for specifying the beginning of the current slide's speaker notes (defaults to `note:`). The `data-charset` attribute is optional and specifies which charset to use when loading the external file. +You can write your content as a separate file and have reveal.js load it at runtime. Note the separator arguments which determine how slides are delimited in the external file. The ```data-charset``` attribute is optional and specifies which charset to use when loading the external file. -When used locally, this feature requires that reveal.js [runs from a local web server](#full-setup). The following example customises all available options: +When used locally, this feature requires that reveal.js [runs from a local web server](#full-setup). ```html -
                -
                ``` @@ -164,19 +92,6 @@ Special syntax (in html comment) is available for adding attributes to the slide
                ``` -#### Configuring *marked* - -We use [marked](https://github.com/chjj/marked) to parse Markdown. To customise marked's rendering, you can pass in options when [configuring Reveal](#configuration): - -```javascript -Reveal.initialize({ - // Options which are passed into marked - // See https://github.com/chjj/marked#options-1 - markdown: { - smartypants: true - } -}); -``` ### Configuration @@ -185,26 +100,12 @@ At the end of your page you need to initialize reveal by running the following c ```javascript Reveal.initialize({ - // Display presentation control arrows + // Display controls in the bottom right corner controls: true, - // Help the user learn the controls by providing hints, for example by - // bouncing the down arrow when they first encounter a vertical slide - controlsTutorial: true, - - // Determines where controls appear, "edges" or "bottom-right" - controlsLayout: 'bottom-right', - - // Visibility rule for backwards navigation arrows; "faded", "hidden" - // or "visible" - controlsBackArrows: 'faded', - // Display a presentation progress bar progress: true, - // Set default timing of 2 minutes per slide - defaultTiming: 120, - // Display the page number of the current slide slideNumber: false, @@ -229,9 +130,6 @@ Reveal.initialize({ // Change the presentation direction to be RTL rtl: false, - // Randomizes the order of slides each time the presentation loads - shuffle: false, - // Turns fragments on and off globally fragments: true, @@ -243,15 +141,6 @@ Reveal.initialize({ // key is pressed help: true, - // Flags if speaker notes should be visible to all viewers - showNotes: false, - - // Global override for autoplaying embedded media (video/audio/iframe) - // - null: Media will only autoplay if data-autoplay is present - // - true: All media will autoplay, regardless of individual setting - // - false: No media will autoplay, regardless of individual setting - autoPlayMedia: null, - // Number of milliseconds between automatically proceeding to the // next slide, disabled when set to 0, this value can be overwritten // by using a data-autoslide attribute on your slides @@ -260,9 +149,6 @@ Reveal.initialize({ // Stop auto-sliding after user input autoSlideStoppable: true, - // Use this method for navigation when auto-sliding - autoSlideMethod: Reveal.navigateNext, - // Enable slide navigation via mouse wheel mouseWheel: false, @@ -270,18 +156,16 @@ Reveal.initialize({ hideAddressBar: true, // Opens links in an iframe preview overlay - // Add `data-preview-link` and `data-preview-link="false"` to customise each link - // individually previewLinks: false, // Transition style - transition: 'slide', // none/fade/slide/convex/concave/zoom + transition: 'default', // none/fade/slide/convex/concave/zoom // Transition speed transitionSpeed: 'default', // default/fast/slow // Transition style for full page slide backgrounds - backgroundTransition: 'fade', // none/fade/slide/convex/concave/zoom + backgroundTransition: 'default', // none/fade/slide/convex/concave/zoom // Number of slides away from the current that are visible viewDistance: 3, @@ -292,14 +176,10 @@ Reveal.initialize({ // Parallax background size parallaxBackgroundSize: '', // CSS syntax, e.g. "2100px 900px" - // Number of pixels to move the parallax background per slide - // - Calculated automatically unless specified - // - Set to 0 to disable movement along an axis - parallaxBackgroundHorizontal: null, - parallaxBackgroundVertical: null, - - // The display mode that will be used to show slides - display: 'block' + // Amount to move parallax background (horizontal and vertical) on slide change + // Number, e.g. 100 + parallaxBackgroundHorizontal: '', + parallaxBackgroundVertical: '' }); ``` @@ -316,6 +196,56 @@ Reveal.configure({ autoSlide: 5000 }); ``` +### Dependencies + +Reveal.js doesn't _rely_ on any third party scripts to work but a few optional libraries are included by default. These libraries are loaded as dependencies in the order they appear, for example: + +```javascript +Reveal.initialize({ + dependencies: [ + // Cross-browser shim that fully implements classList - https://github.com/eligrey/classList.js/ + { src: 'lib/js/classList.js', condition: function() { return !document.body.classList; } }, + + // Interpret Markdown in
                elements + { src: 'plugin/markdown/marked.js', condition: function() { return !!document.querySelector( '[data-markdown]' ); } }, + { src: 'plugin/markdown/markdown.js', condition: function() { return !!document.querySelector( '[data-markdown]' ); } }, + + // Syntax highlight for elements + { src: 'plugin/highlight/highlight.js', async: true, callback: function() { hljs.initHighlightingOnLoad(); } }, + + // Zoom in and out with Alt+click + { src: 'plugin/zoom-js/zoom.js', async: true }, + + // Speaker notes + { src: 'plugin/notes/notes.js', async: true }, + + // Remote control your reveal.js presentation using a touch device + { src: 'plugin/remotes/remotes.js', async: true }, + + // MathJax + { src: 'plugin/math/math.js', async: true } + ] +}); +``` + +You can add your own extensions using the same syntax. The following properties are available for each dependency object: +- **src**: Path to the script to load +- **async**: [optional] Flags if the script should load after reveal.js has started, defaults to false +- **callback**: [optional] Function to execute when the script has loaded +- **condition**: [optional] Function which must return true for the script to be loaded + + +### Ready Event + +A 'ready' event is fired when reveal.js has loaded all non-async dependencies and is ready to start navigating. To check if reveal.js is already 'ready' you can call `Reveal.isReady()`. + +```javascript +Reveal.addEventListener( 'ready', function( event ) { + // event.currentSlide, event.indexh, event.indexv +} ); +``` + + ### Presentation Size All presentations have a normal size, that is the resolution at which they are authored. The framework will automatically scale presentations uniformly based on this size to ensure that everything fits on any given display or viewport. @@ -343,69 +273,6 @@ Reveal.initialize({ }); ``` -If you wish to disable this behavior and do your own scaling (e.g. using media queries), try these settings: - -```javascript -Reveal.initialize({ - - ... - - width: "100%", - height: "100%", - margin: 0, - minScale: 1, - maxScale: 1 -}); -``` - -### Dependencies - -Reveal.js doesn't _rely_ on any third party scripts to work but a few optional libraries are included by default. These libraries are loaded as dependencies in the order they appear, for example: - -```javascript -Reveal.initialize({ - dependencies: [ - // Cross-browser shim that fully implements classList - https://github.com/eligrey/classList.js/ - { src: 'lib/js/classList.js', condition: function() { return !document.body.classList; } }, - - // Interpret Markdown in
                elements - { src: 'plugin/markdown/marked.js', condition: function() { return !!document.querySelector( '[data-markdown]' ); } }, - { src: 'plugin/markdown/markdown.js', condition: function() { return !!document.querySelector( '[data-markdown]' ); } }, - - // Syntax highlight for elements - { src: 'plugin/highlight/highlight.js', async: true, callback: function() { hljs.initHighlightingOnLoad(); } }, - - // Zoom in and out with Alt+click - { src: 'plugin/zoom-js/zoom.js', async: true }, - - // Speaker notes - { src: 'plugin/notes/notes.js', async: true }, - - // MathJax - { src: 'plugin/math/math.js', async: true } - ] -}); -``` - -You can add your own extensions using the same syntax. The following properties are available for each dependency object: -- **src**: Path to the script to load -- **async**: [optional] Flags if the script should load after reveal.js has started, defaults to false -- **callback**: [optional] Function to execute when the script has loaded -- **condition**: [optional] Function which must return true for the script to be loaded - -To load these dependencies, reveal.js requires [head.js](http://headjs.com/) *(a script loading library)* to be loaded before reveal.js. - -### Ready Event - -A 'ready' event is fired when reveal.js has loaded all non-async dependencies and is ready to start navigating. To check if reveal.js is already 'ready' you can call `Reveal.isReady()`. - -```javascript -Reveal.addEventListener( 'ready', function( event ) { - // event.currentSlide, event.indexh, event.indexv -} ); -``` - -Note that we also add a `.ready` class to the `.reveal` element so that you can hook into this with CSS. ### Auto-sliding @@ -429,8 +296,6 @@ You can also override the slide duration for individual slides and fragments by
                ``` -To override the method used for navigation when auto-sliding, you can specify the ```autoSlideMethod``` setting. To only navigate along the top layer and ignore vertical slides, set this to ```Reveal.navigateRight```. - Whenever the auto-slide mode is resumed or paused the ```autoslideresumed``` and ```autoslidepaused``` events are fired. @@ -448,13 +313,6 @@ Reveal.configure({ }); ``` -### Touch Navigation - -You can swipe to navigate through a presentation on any touch-enabled device. Horizontal swipes change between horizontal slides, vertical swipes change between vertical slides. If you wish to disable this you can set the `touch` config option to false when initializing reveal.js. - -If there's some part of your content that needs to remain accessible to touch events you'll need to highlight this by adding a `data-prevent-swipe` attribute to the element. One common example where this is useful is elements that need to be scrolled. - - ### Lazy Loading When working on presentation with a lot of media or iframe content it's important to load lazily. Lazy loading means that reveal.js will only load content for the few slides nearest to the current slide. The number of slides that are preloaded is determined by the `viewDistance` configuration option. @@ -489,18 +347,11 @@ Reveal.next(); Reveal.prevFragment(); Reveal.nextFragment(); -// Randomize the order of slides -Reveal.shuffle(); - // Toggle presentation states, optionally pass true/false to force on/off Reveal.toggleOverview(); Reveal.togglePause(); Reveal.toggleAutoSlide(); -// Shows a help overlay with keyboard shortcuts, optionally pass true/false -// to force on/off -Reveal.toggleHelp(); - // Change a config value at runtime Reveal.configure({ controls: true }); @@ -514,14 +365,9 @@ Reveal.getScale(); Reveal.getPreviousSlide(); Reveal.getCurrentSlide(); -Reveal.getIndices(); // { h: 0, v: 0 } } -Reveal.getPastSlideCount(); -Reveal.getProgress(); // (0 == first slide, 1 == last slide) -Reveal.getSlides(); // Array of all slides -Reveal.getTotalSlides(); // total number of slides - -// Returns the speaker notes for the current slide -Reveal.getSlideNotes(); +Reveal.getIndices(); // { h: 0, v: 0 } } +Reveal.getProgress(); // 0-1 +Reveal.getTotalSlides(); // State checks Reveal.isFirstSlide(); @@ -574,59 +420,26 @@ Reveal.addEventListener( 'somestate', function() { ### Slide Backgrounds -Slides are contained within a limited portion of the screen by default to allow them to fit any display and scale uniformly. You can apply full page backgrounds outside of the slide area by adding a ```data-background``` attribute to your ```
                ``` elements. Four different types of backgrounds are supported: color, image, video and iframe. +Slides are contained within a limited portion of the screen by default to allow them to fit any display and scale uniformly. You can apply full page backgrounds outside of the slide area by adding a ```data-background``` attribute to your ```
                ``` elements. Four different types of backgrounds are supported: color, image, video and iframe. Below are a few examples. -#### Color Backgrounds -All CSS color formats are supported, like rgba() or hsl(). ```html -
                -

                Color

                +
                +

                All CSS color formats are supported, like rgba() or hsl().

                +
                +
                +

                This slide will have a full-size background image.

                +
                +
                +

                This background image will be sized to 100px and repeated.

                +
                +
                +

                Video. Multiple sources can be defined using a comma separated list. Video will loop when the data-background-video-loop attribute is provided.

                +
                +
                +

                Embeds a web page as a background. Note that the page won't be interactive.

                ``` -#### Image Backgrounds -By default, background images are resized to cover the full page. Available options: - -| Attribute | Default | Description | -| :--------------------------- | :--------- | :---------- | -| data-background-image | | URL of the image to show. GIFs restart when the slide opens. | -| data-background-size | cover | See [background-size](https://developer.mozilla.org/docs/Web/CSS/background-size) on MDN. | -| data-background-position | center | See [background-position](https://developer.mozilla.org/docs/Web/CSS/background-position) on MDN. | -| data-background-repeat | no-repeat | See [background-repeat](https://developer.mozilla.org/docs/Web/CSS/background-repeat) on MDN. | -```html -
                -

                Image

                -
                -
                -

                This background image will be sized to 100px and repeated

                -
                -``` - -#### Video Backgrounds -Automatically plays a full size video behind the slide. - -| Attribute | Default | Description | -| :--------------------------- | :------ | :---------- | -| data-background-video | | A single video source, or a comma separated list of video sources. | -| data-background-video-loop | false | Flags if the video should play repeatedly. | -| data-background-video-muted | false | Flags if the audio should be muted. | -| data-background-size | cover | Use `cover` for full screen and some cropping or `contain` for letterboxing. | - -```html -
                -

                Video

                -
                -``` - -#### Iframe Backgrounds -Embeds a web page as a slide background that covers 100% of the reveal.js width and height. The iframe is in the background layer, behind your slides, and as such it's not possible to interact with it by default. To make your background interactive, you can add the `data-background-interactive` attribute. -```html -
                -

                Iframe

                -
                -``` - -#### Background Transitions Backgrounds transition using a fade animation by default. This can be changed to a linear sliding transition by passing ```backgroundTransition: 'slide'``` to the ```Reveal.initialize()``` call. Alternatively you can set ```data-background-transition``` on any section with a background to override that specific transition. @@ -643,16 +456,16 @@ Reveal.initialize({ // Parallax background size parallaxBackgroundSize: '', // CSS syntax, e.g. "2100px 900px" - currently only pixels are supported (don't use % or auto) - // Number of pixels to move the parallax background per slide - // - Calculated automatically unless specified - // - Set to 0 to disable movement along an axis + // Amount of pixels to move the parallax background per slide step, + // a value of 0 disables movement along the given axis + // These are optional, if they aren't specified they'll be calculated automatically parallaxBackgroundHorizontal: 200, parallaxBackgroundVertical: 50 }); ``` -Make sure that the background size is much bigger than screen size to allow for some scrolling. [View example](http://revealjs.com/?parallaxBackgroundImage=https%3A%2F%2Fs3.amazonaws.com%2Fhakim-static%2Freveal-js%2Freveal-parallax-1.jpg¶llaxBackgroundSize=2100px%20900px). +Make sure that the background size is much bigger than screen size to allow for some scrolling. [View example](http://lab.hakim.se/reveal-js/?parallaxBackgroundImage=https%3A%2F%2Fs3.amazonaws.com%2Fhakim-static%2Freveal-js%2Freveal-parallax-1.jpg¶llaxBackgroundSize=2100px%20900px). @@ -673,15 +486,15 @@ You can also use different in and out transitions for the same slide: ```html
                - The train goes on … + The train goes on …
                -
                - and on … +
                + and on …
                -
                +
                and stops.
                -
                +
                (Passengers entering and leaving)
                @@ -690,6 +503,9 @@ You can also use different in and out transitions for the same slide: ``` +Note that this does not work with the page and cube transitions. + + ### Internal links It's easy to link between slides. The first example below targets the index of another slide whereas the second targets a slide with an ID attribute (```
                ```): @@ -712,7 +528,7 @@ You can also add relative navigation links, similar to the built in reveal.js co ### Fragments -Fragments are used to highlight individual elements on a slide. Every element with the class ```fragment``` will be stepped through before moving on to the next slide. Here's an example: http://revealjs.com/#/fragments +Fragments are used to highlight individual elements on a slide. Every element with the class ```fragment``` will be stepped through before moving on to the next slide. Here's an example: http://lab.hakim.se/reveal-js/#/fragments The default fragment style is to start out invisible and fade in. This style can be changed by appending a different class to the fragment: @@ -721,7 +537,6 @@ The default fragment style is to start out invisible and fade in. This style can

                grow

                shrink

                fade-out

                -

                fade-up (also down, left and right!)

                visible only once

                blue only once

                highlight-red

                @@ -767,41 +582,33 @@ Reveal.addEventListener( 'fragmenthidden', function( event ) { ### Code syntax highlighting -By default, Reveal is configured with [highlight.js](https://highlightjs.org/) for code syntax highlighting. To enable syntax highlighting, you'll have to load the highlight plugin ([plugin/highlight/highlight.js](plugin/highlight/highlight.js)) and a highlight.js CSS theme (Reveal comes packaged with the zenburn theme: [lib/css/zenburn.css](lib/css/zenburn.css)). - -Below is an example with clojure code that will be syntax highlighted. When the `data-trim` attribute is present, surrounding whitespace is automatically removed. HTML will be escaped by default. To avoid this, for example if you are using `` to call out a line of code, add the `data-noescape` attribute to the `` element. +By default, Reveal is configured with [highlight.js](http://softwaremaniacs.org/soft/highlight/en/) for code syntax highlighting. Below is an example with clojure code that will be syntax highlighted. When the `data-trim` attribute is present surrounding whitespace is automatically removed. ```html
                -
                
                +	
                
                 (def lazy-fib
                   (concat
                    [0 1]
                -   ((fn rfib [a b]
                +   ((fn rfib [a b]
                         (lazy-cons (+ a b) (rfib b (+ a b)))) 0 1)))
                 	
                ``` ### Slide number -If you would like to display the page number of the current slide you can do so using the ```slideNumber``` and ```showSlideNumber``` configuration values. +If you would like to display the page number of the current slide you can do so using the ```slideNumber``` configuration value. ```javascript // Shows the slide number using default formatting Reveal.configure({ slideNumber: true }); // Slide number formatting can be configured using these variables: -// "h.v": horizontal . vertical slide number (default) -// "h/v": horizontal / vertical slide number -// "c": flattened slide number -// "c/t": flattened slide number / total slides -Reveal.configure({ slideNumber: 'c/t' }); - -// Control which views the slide number displays on using the "showSlideNumber" value: -// "all": show on all views (default) -// "speaker": only show slide numbers on speaker notes view -// "print": only show slide numbers when printing to PDF -Reveal.configure({ showSlideNumber: 'speaker' }); +// h: current slide's horizontal index +// v: current slide's vertical index +// c: current slide index (flattened) +// t: total number of slides (flattened) +Reveal.configure({ slideNumber: 'c / t' }); ``` @@ -819,26 +626,20 @@ Reveal.addEventListener( 'overviewhidden', function( event ) { /* ... */ } ); Reveal.toggleOverview(); ``` - ### Fullscreen mode Just press »F« on your keyboard to show your presentation in fullscreen mode. Press the »ESC« key to exit fullscreen mode. ### Embedded media +Embedded HTML5 `
                diff --git a/doc/pub/Splines/html/reveal.js/plugin/markdown/example.md b/doc/pub/Splines/html/reveal.js/plugin/markdown/example.md index 89c75345e..6f6f577a1 100644 --- a/doc/pub/Splines/html/reveal.js/plugin/markdown/example.md +++ b/doc/pub/Splines/html/reveal.js/plugin/markdown/example.md @@ -29,8 +29,3 @@ Content 3.1 ## External 3.2 Content 3.2 - - -## External 3.3 - -![External Image](https://s3.amazonaws.com/static.slid.es/logo/v2/slides-symbol-512x512.png) diff --git a/doc/pub/Splines/html/reveal.js/plugin/markdown/markdown.js b/doc/pub/Splines/html/reveal.js/plugin/markdown/markdown.js index aa08ee5ed..15e3b40b3 100644 --- a/doc/pub/Splines/html/reveal.js/plugin/markdown/markdown.js +++ b/doc/pub/Splines/html/reveal.js/plugin/markdown/markdown.js @@ -4,26 +4,33 @@ * of external markdown documents. */ (function( root, factory ) { - if (typeof define === 'function' && define.amd) { - root.marked = require( './marked' ); - root.RevealMarkdown = factory( root.marked ); - root.RevealMarkdown.initialize(); - } else if( typeof exports === 'object' ) { + if( typeof exports === 'object' ) { module.exports = factory( require( './marked' ) ); - } else { + } + else { // Browser globals (root is window) root.RevealMarkdown = factory( root.marked ); root.RevealMarkdown.initialize(); } }( this, function( marked ) { + if( typeof marked === 'undefined' ) { + throw 'The reveal.js Markdown plugin requires marked to be loaded'; + } + + if( typeof hljs !== 'undefined' ) { + marked.setOptions({ + highlight: function( lang, code ) { + return hljs.highlightAuto( lang, code ).value; + } + }); + } + var DEFAULT_SLIDE_SEPARATOR = '^\r?\n---\r?\n$', - DEFAULT_NOTES_SEPARATOR = 'notes?:', + DEFAULT_NOTES_SEPARATOR = 'note:', DEFAULT_ELEMENT_ATTRIBUTES_SEPARATOR = '\\\.element\\\s*?(.+?)$', DEFAULT_SLIDE_ATTRIBUTES_SEPARATOR = '\\\.slide:\\\s*?(\\\S.+?)$'; - var SCRIPT_END_PLACEHOLDER = '__SCRIPT_END__'; - /** * Retrieves the markdown contents of a slide section @@ -31,15 +38,11 @@ */ function getMarkdownFromSlide( section ) { - // look for a ' ); - var leadingWs = text.match( /^\n?(\s*)/ )[1].length, leadingTabs = text.match( /^\n?(\t*)/ )[1].length; @@ -109,13 +112,9 @@ var notesMatch = content.split( new RegExp( options.notesSeparator, 'mgi' ) ); if( notesMatch.length === 2 ) { - content = notesMatch[0] + ''; + content = notesMatch[0] + ''; } - // prevent script end tags in the content from interfering - // with parsing - content = content.replace( /<\/script>/g, SCRIPT_END_PLACEHOLDER ); - return ''; } @@ -178,7 +177,7 @@ markdownSections += '
                '; sectionStack[i].forEach( function( child ) { - markdownSections += '
                ' + createMarkdownSlide( child, options ) + '
                '; + markdownSections += '
                ' + createMarkdownSlide( child, options ) + '
                '; } ); markdownSections += '
                '; @@ -380,24 +379,6 @@ return { initialize: function() { - if( typeof marked === 'undefined' ) { - throw 'The reveal.js Markdown plugin requires marked to be loaded'; - } - - if( typeof hljs !== 'undefined' ) { - marked.setOptions({ - highlight: function( code, lang ) { - return hljs.highlightAuto( code, [lang] ).value; - } - }); - } - - var options = Reveal.getConfig().markdown; - - if ( options ) { - marked.setOptions( options ); - } - processSlides(); convertSlides(); }, diff --git a/doc/pub/Splines/html/reveal.js/plugin/markdown/marked.js b/doc/pub/Splines/html/reveal.js/plugin/markdown/marked.js index 555c1dc1d..70af29bf9 100644 --- a/doc/pub/Splines/html/reveal.js/plugin/markdown/marked.js +++ b/doc/pub/Splines/html/reveal.js/plugin/markdown/marked.js @@ -3,4 +3,4 @@ * Copyright (c) 2011-2014, Christopher Jeffrey. (MIT Licensed) * https://github.com/chjj/marked */ -(function(){var block={newline:/^\n+/,code:/^( {4}[^\n]+\n*)+/,fences:noop,hr:/^( *[-*_]){3,} *(?:\n+|$)/,heading:/^ *(#{1,6}) *([^\n]+?) *#* *(?:\n+|$)/,nptable:noop,lheading:/^([^\n]+)\n *(=|-){2,} *(?:\n+|$)/,blockquote:/^( *>[^\n]+(\n(?!def)[^\n]+)*\n*)+/,list:/^( *)(bull) [\s\S]+?(?:hr|def|\n{2,}(?! )(?!\1bull )\n*|\s*$)/,html:/^ *(?:comment *(?:\n|\s*$)|closed *(?:\n{2,}|\s*$)|closing *(?:\n{2,}|\s*$))/,def:/^ *\[([^\]]+)\]: *]+)>?(?: +["(]([^\n]+)[")])? *(?:\n+|$)/,table:noop,paragraph:/^((?:[^\n]+\n?(?!hr|heading|lheading|blockquote|tag|def))+)\n*/,text:/^[^\n]+/};block.bullet=/(?:[*+-]|\d+\.)/;block.item=/^( *)(bull) [^\n]*(?:\n(?!\1bull )[^\n]*)*/;block.item=replace(block.item,"gm")(/bull/g,block.bullet)();block.list=replace(block.list)(/bull/g,block.bullet)("hr","\\n+(?=\\1?(?:[-*_] *){3,}(?:\\n+|$))")("def","\\n+(?="+block.def.source+")")();block.blockquote=replace(block.blockquote)("def",block.def)();block._tag="(?!(?:"+"a|em|strong|small|s|cite|q|dfn|abbr|data|time|code"+"|var|samp|kbd|sub|sup|i|b|u|mark|ruby|rt|rp|bdi|bdo"+"|span|br|wbr|ins|del|img)\\b)\\w+(?!:/|[^\\w\\s@]*@)\\b";block.html=replace(block.html)("comment",//)("closed",/<(tag)[\s\S]+?<\/\1>/)("closing",/])*?>/)(/tag/g,block._tag)();block.paragraph=replace(block.paragraph)("hr",block.hr)("heading",block.heading)("lheading",block.lheading)("blockquote",block.blockquote)("tag","<"+block._tag)("def",block.def)();block.normal=merge({},block);block.gfm=merge({},block.normal,{fences:/^ *(`{3,}|~{3,})[ \.]*(\S+)? *\n([\s\S]*?)\s*\1 *(?:\n+|$)/,paragraph:/^/,heading:/^ *(#{1,6}) +([^\n]+?) *#* *(?:\n+|$)/});block.gfm.paragraph=replace(block.paragraph)("(?!","(?!"+block.gfm.fences.source.replace("\\1","\\2")+"|"+block.list.source.replace("\\1","\\3")+"|")();block.tables=merge({},block.gfm,{nptable:/^ *(\S.*\|.*)\n *([-:]+ *\|[-| :]*)\n((?:.*\|.*(?:\n|$))*)\n*/,table:/^ *\|(.+)\n *\|( *[-:]+[-| :]*)\n((?: *\|.*(?:\n|$))*)\n*/});function Lexer(options){this.tokens=[];this.tokens.links={};this.options=options||marked.defaults;this.rules=block.normal;if(this.options.gfm){if(this.options.tables){this.rules=block.tables}else{this.rules=block.gfm}}}Lexer.rules=block;Lexer.lex=function(src,options){var lexer=new Lexer(options);return lexer.lex(src)};Lexer.prototype.lex=function(src){src=src.replace(/\r\n|\r/g,"\n").replace(/\t/g," ").replace(/\u00a0/g," ").replace(/\u2424/g,"\n");return this.token(src,true)};Lexer.prototype.token=function(src,top,bq){var src=src.replace(/^ +$/gm,""),next,loose,cap,bull,b,item,space,i,l;while(src){if(cap=this.rules.newline.exec(src)){src=src.substring(cap[0].length);if(cap[0].length>1){this.tokens.push({type:"space"})}}if(cap=this.rules.code.exec(src)){src=src.substring(cap[0].length);cap=cap[0].replace(/^ {4}/gm,"");this.tokens.push({type:"code",text:!this.options.pedantic?cap.replace(/\n+$/,""):cap});continue}if(cap=this.rules.fences.exec(src)){src=src.substring(cap[0].length);this.tokens.push({type:"code",lang:cap[2],text:cap[3]||""});continue}if(cap=this.rules.heading.exec(src)){src=src.substring(cap[0].length);this.tokens.push({type:"heading",depth:cap[1].length,text:cap[2]});continue}if(top&&(cap=this.rules.nptable.exec(src))){src=src.substring(cap[0].length);item={type:"table",header:cap[1].replace(/^ *| *\| *$/g,"").split(/ *\| */),align:cap[2].replace(/^ *|\| *$/g,"").split(/ *\| */),cells:cap[3].replace(/\n$/,"").split("\n")};for(i=0;i ?/gm,"");this.token(cap,top,true);this.tokens.push({type:"blockquote_end"});continue}if(cap=this.rules.list.exec(src)){src=src.substring(cap[0].length);bull=cap[2];this.tokens.push({type:"list_start",ordered:bull.length>1});cap=cap[0].match(this.rules.item);next=false;l=cap.length;i=0;for(;i1&&b.length>1)){src=cap.slice(i+1).join("\n")+src;i=l-1}}loose=next||/\n\n(?!\s*$)/.test(item);if(i!==l-1){next=item.charAt(item.length-1)==="\n";if(!loose)loose=next}this.tokens.push({type:loose?"loose_item_start":"list_item_start"});this.token(item,false,bq);this.tokens.push({type:"list_item_end"})}this.tokens.push({type:"list_end"});continue}if(cap=this.rules.html.exec(src)){src=src.substring(cap[0].length);this.tokens.push({type:this.options.sanitize?"paragraph":"html",pre:!this.options.sanitizer&&(cap[1]==="pre"||cap[1]==="script"||cap[1]==="style"),text:cap[0]});continue}if(!bq&&top&&(cap=this.rules.def.exec(src))){src=src.substring(cap[0].length);this.tokens.links[cap[1].toLowerCase()]={href:cap[2],title:cap[3]};continue}if(top&&(cap=this.rules.table.exec(src))){src=src.substring(cap[0].length);item={type:"table",header:cap[1].replace(/^ *| *\| *$/g,"").split(/ *\| */),align:cap[2].replace(/^ *|\| *$/g,"").split(/ *\| */),cells:cap[3].replace(/(?: *\| *)?\n$/,"").split("\n")};for(i=0;i])/,autolink:/^<([^ >]+(@|:\/)[^ >]+)>/,url:noop,tag:/^|^<\/?\w+(?:"[^"]*"|'[^']*'|[^'">])*?>/,link:/^!?\[(inside)\]\(href\)/,reflink:/^!?\[(inside)\]\s*\[([^\]]*)\]/,nolink:/^!?\[((?:\[[^\]]*\]|[^\[\]])*)\]/,strong:/^__([\s\S]+?)__(?!_)|^\*\*([\s\S]+?)\*\*(?!\*)/,em:/^\b_((?:[^_]|__)+?)_\b|^\*((?:\*\*|[\s\S])+?)\*(?!\*)/,code:/^(`+)\s*([\s\S]*?[^`])\s*\1(?!`)/,br:/^ {2,}\n(?!\s*$)/,del:noop,text:/^[\s\S]+?(?=[\\?(?:\s+['"]([\s\S]*?)['"])?\s*/;inline.link=replace(inline.link)("inside",inline._inside)("href",inline._href)();inline.reflink=replace(inline.reflink)("inside",inline._inside)();inline.normal=merge({},inline);inline.pedantic=merge({},inline.normal,{strong:/^__(?=\S)([\s\S]*?\S)__(?!_)|^\*\*(?=\S)([\s\S]*?\S)\*\*(?!\*)/,em:/^_(?=\S)([\s\S]*?\S)_(?!_)|^\*(?=\S)([\s\S]*?\S)\*(?!\*)/});inline.gfm=merge({},inline.normal,{escape:replace(inline.escape)("])","~|])")(),url:/^(https?:\/\/[^\s<]+[^<.,:;"')\]\s])/,del:/^~~(?=\S)([\s\S]*?\S)~~/,text:replace(inline.text)("]|","~]|")("|","|https?://|")()});inline.breaks=merge({},inline.gfm,{br:replace(inline.br)("{2,}","*")(),text:replace(inline.gfm.text)("{2,}","*")()});function InlineLexer(links,options){this.options=options||marked.defaults;this.links=links;this.rules=inline.normal;this.renderer=this.options.renderer||new Renderer;this.renderer.options=this.options;if(!this.links){throw new Error("Tokens array requires a `links` property.")}if(this.options.gfm){if(this.options.breaks){this.rules=inline.breaks}else{this.rules=inline.gfm}}else if(this.options.pedantic){this.rules=inline.pedantic}}InlineLexer.rules=inline;InlineLexer.output=function(src,links,options){var inline=new InlineLexer(links,options);return inline.output(src)};InlineLexer.prototype.output=function(src){var out="",link,text,href,cap;while(src){if(cap=this.rules.escape.exec(src)){src=src.substring(cap[0].length);out+=cap[1];continue}if(cap=this.rules.autolink.exec(src)){src=src.substring(cap[0].length);if(cap[2]==="@"){text=cap[1].charAt(6)===":"?this.mangle(cap[1].substring(7)):this.mangle(cap[1]);href=this.mangle("mailto:")+text}else{text=escape(cap[1]);href=text}out+=this.renderer.link(href,null,text);continue}if(!this.inLink&&(cap=this.rules.url.exec(src))){src=src.substring(cap[0].length);text=escape(cap[1]);href=text;out+=this.renderer.link(href,null,text);continue}if(cap=this.rules.tag.exec(src)){if(!this.inLink&&/^/i.test(cap[0])){this.inLink=false}src=src.substring(cap[0].length);out+=this.options.sanitize?this.options.sanitizer?this.options.sanitizer(cap[0]):escape(cap[0]):cap[0];continue}if(cap=this.rules.link.exec(src)){src=src.substring(cap[0].length);this.inLink=true;out+=this.outputLink(cap,{href:cap[2],title:cap[3]});this.inLink=false;continue}if((cap=this.rules.reflink.exec(src))||(cap=this.rules.nolink.exec(src))){src=src.substring(cap[0].length);link=(cap[2]||cap[1]).replace(/\s+/g," ");link=this.links[link.toLowerCase()];if(!link||!link.href){out+=cap[0].charAt(0);src=cap[0].substring(1)+src;continue}this.inLink=true;out+=this.outputLink(cap,link);this.inLink=false;continue}if(cap=this.rules.strong.exec(src)){src=src.substring(cap[0].length);out+=this.renderer.strong(this.output(cap[2]||cap[1]));continue}if(cap=this.rules.em.exec(src)){src=src.substring(cap[0].length);out+=this.renderer.em(this.output(cap[2]||cap[1]));continue}if(cap=this.rules.code.exec(src)){src=src.substring(cap[0].length);out+=this.renderer.codespan(escape(cap[2],true));continue}if(cap=this.rules.br.exec(src)){src=src.substring(cap[0].length);out+=this.renderer.br();continue}if(cap=this.rules.del.exec(src)){src=src.substring(cap[0].length);out+=this.renderer.del(this.output(cap[1]));continue}if(cap=this.rules.text.exec(src)){src=src.substring(cap[0].length);out+=this.renderer.text(escape(this.smartypants(cap[0])));continue}if(src){throw new Error("Infinite loop on byte: "+src.charCodeAt(0))}}return out};InlineLexer.prototype.outputLink=function(cap,link){var href=escape(link.href),title=link.title?escape(link.title):null;return cap[0].charAt(0)!=="!"?this.renderer.link(href,title,this.output(cap[1])):this.renderer.image(href,title,escape(cap[1]))};InlineLexer.prototype.smartypants=function(text){if(!this.options.smartypants)return text;return text.replace(/---/g,"—").replace(/--/g,"–").replace(/(^|[-\u2014/(\[{"\s])'/g,"$1‘").replace(/'/g,"’").replace(/(^|[-\u2014/(\[{\u2018\s])"/g,"$1“").replace(/"/g,"”").replace(/\.{3}/g,"…")};InlineLexer.prototype.mangle=function(text){if(!this.options.mangle)return text;var out="",l=text.length,i=0,ch;for(;i.5){ch="x"+ch.toString(16)}out+="&#"+ch+";"}return out};function Renderer(options){this.options=options||{}}Renderer.prototype.code=function(code,lang,escaped){if(this.options.highlight){var out=this.options.highlight(code,lang);if(out!=null&&out!==code){escaped=true;code=out}}if(!lang){return"
                "+(escaped?code:escape(code,true))+"\n
                "}return'
                '+(escaped?code:escape(code,true))+"\n
                \n"};Renderer.prototype.blockquote=function(quote){return"
                \n"+quote+"
                \n"};Renderer.prototype.html=function(html){return html};Renderer.prototype.heading=function(text,level,raw){return"'+text+"\n"};Renderer.prototype.hr=function(){return this.options.xhtml?"
                \n":"
                \n"};Renderer.prototype.list=function(body,ordered){var type=ordered?"ol":"ul";return"<"+type+">\n"+body+"\n"};Renderer.prototype.listitem=function(text){return"
              • "+text+"
              • \n"};Renderer.prototype.paragraph=function(text){return"

                "+text+"

                \n"};Renderer.prototype.table=function(header,body){return"\n"+"\n"+header+"\n"+"\n"+body+"\n"+"
                \n"};Renderer.prototype.tablerow=function(content){return"\n"+content+"\n"};Renderer.prototype.tablecell=function(content,flags){var type=flags.header?"th":"td";var tag=flags.align?"<"+type+' style="text-align:'+flags.align+'">':"<"+type+">";return tag+content+"\n"};Renderer.prototype.strong=function(text){return""+text+""};Renderer.prototype.em=function(text){return""+text+""};Renderer.prototype.codespan=function(text){return""+text+""};Renderer.prototype.br=function(){return this.options.xhtml?"
                ":"
                "};Renderer.prototype.del=function(text){return""+text+""};Renderer.prototype.link=function(href,title,text){if(this.options.sanitize){try{var prot=decodeURIComponent(unescape(href)).replace(/[^\w:]/g,"").toLowerCase()}catch(e){return""}if(prot.indexOf("javascript:")===0||prot.indexOf("vbscript:")===0){return""}}var out='
                ";return out};Renderer.prototype.image=function(href,title,text){var out=''+text+'":">";return out};Renderer.prototype.text=function(text){return text};function Parser(options){this.tokens=[];this.token=null;this.options=options||marked.defaults;this.options.renderer=this.options.renderer||new Renderer;this.renderer=this.options.renderer;this.renderer.options=this.options}Parser.parse=function(src,options,renderer){var parser=new Parser(options,renderer);return parser.parse(src)};Parser.prototype.parse=function(src){this.inline=new InlineLexer(src.links,this.options,this.renderer);this.tokens=src.reverse();var out="";while(this.next()){out+=this.tok()}return out};Parser.prototype.next=function(){return this.token=this.tokens.pop()};Parser.prototype.peek=function(){return this.tokens[this.tokens.length-1]||0};Parser.prototype.parseText=function(){var body=this.token.text;while(this.peek().type==="text"){body+="\n"+this.next().text}return this.inline.output(body)};Parser.prototype.tok=function(){switch(this.token.type){case"space":{return""}case"hr":{return this.renderer.hr()}case"heading":{return this.renderer.heading(this.inline.output(this.token.text),this.token.depth,this.token.text)}case"code":{return this.renderer.code(this.token.text,this.token.lang,this.token.escaped)}case"table":{var header="",body="",i,row,cell,flags,j;cell="";for(i=0;i/g,">").replace(/"/g,""").replace(/'/g,"'")}function unescape(html){return html.replace(/&([#\w]+);/g,function(_,n){n=n.toLowerCase();if(n==="colon")return":";if(n.charAt(0)==="#"){return n.charAt(1)==="x"?String.fromCharCode(parseInt(n.substring(2),16)):String.fromCharCode(+n.substring(1))}return""})}function replace(regex,opt){regex=regex.source;opt=opt||"";return function self(name,val){if(!name)return new RegExp(regex,opt);val=val.source||val;val=val.replace(/(^|[^\[])\^/g,"$1");regex=regex.replace(name,val);return self}}function noop(){}noop.exec=noop;function merge(obj){var i=1,target,key;for(;iAn error occured:

                "+escape(e.message+"",true)+"
                "}throw e}}marked.options=marked.setOptions=function(opt){merge(marked.defaults,opt);return marked};marked.defaults={gfm:true,tables:true,breaks:false,pedantic:false,sanitize:false,sanitizer:null,mangle:true,smartLists:false,silent:false,highlight:null,langPrefix:"lang-",smartypants:false,headerPrefix:"",renderer:new Renderer,xhtml:false};marked.Parser=Parser;marked.parser=Parser.parse;marked.Renderer=Renderer;marked.Lexer=Lexer;marked.lexer=Lexer.lex;marked.InlineLexer=InlineLexer;marked.inlineLexer=InlineLexer.output;marked.parse=marked;if(typeof module!=="undefined"&&typeof exports==="object"){module.exports=marked}else if(typeof define==="function"&&define.amd){define(function(){return marked})}else{this.marked=marked}}).call(function(){return this||(typeof window!=="undefined"?window:global)}()); \ No newline at end of file +(function(){function e(e){this.tokens=[],this.tokens.links={},this.options=e||a.defaults,this.rules=p.normal,this.options.gfm&&(this.rules=this.options.tables?p.tables:p.gfm)}function t(e,t){if(this.options=t||a.defaults,this.links=e,this.rules=u.normal,this.renderer=this.options.renderer||new n,this.renderer.options=this.options,!this.links)throw new Error("Tokens array requires a `links` property.");this.options.gfm?this.rules=this.options.breaks?u.breaks:u.gfm:this.options.pedantic&&(this.rules=u.pedantic)}function n(e){this.options=e||{}}function r(e){this.tokens=[],this.token=null,this.options=e||a.defaults,this.options.renderer=this.options.renderer||new n,this.renderer=this.options.renderer,this.renderer.options=this.options}function s(e,t){return e.replace(t?/&/g:/&(?!#?\w+;)/g,"&").replace(//g,">").replace(/"/g,""").replace(/'/g,"'")}function i(e){return e.replace(/&([#\w]+);/g,function(e,t){return t=t.toLowerCase(),"colon"===t?":":"#"===t.charAt(0)?String.fromCharCode("x"===t.charAt(1)?parseInt(t.substring(2),16):+t.substring(1)):""})}function l(e,t){return e=e.source,t=t||"",function n(r,s){return r?(s=s.source||s,s=s.replace(/(^|[^\[])\^/g,"$1"),e=e.replace(r,s),n):new RegExp(e,t)}}function o(){}function h(e){for(var t,n,r=1;rAn error occured:

                "+s(c.message+"",!0)+"
                ";throw c}}var p={newline:/^\n+/,code:/^( {4}[^\n]+\n*)+/,fences:o,hr:/^( *[-*_]){3,} *(?:\n+|$)/,heading:/^ *(#{1,6}) *([^\n]+?) *#* *(?:\n+|$)/,nptable:o,lheading:/^([^\n]+)\n *(=|-){2,} *(?:\n+|$)/,blockquote:/^( *>[^\n]+(\n(?!def)[^\n]+)*\n*)+/,list:/^( *)(bull) [\s\S]+?(?:hr|def|\n{2,}(?! )(?!\1bull )\n*|\s*$)/,html:/^ *(?:comment *(?:\n|\s*$)|closed *(?:\n{2,}|\s*$)|closing *(?:\n{2,}|\s*$))/,def:/^ *\[([^\]]+)\]: *]+)>?(?: +["(]([^\n]+)[")])? *(?:\n+|$)/,table:o,paragraph:/^((?:[^\n]+\n?(?!hr|heading|lheading|blockquote|tag|def))+)\n*/,text:/^[^\n]+/};p.bullet=/(?:[*+-]|\d+\.)/,p.item=/^( *)(bull) [^\n]*(?:\n(?!\1bull )[^\n]*)*/,p.item=l(p.item,"gm")(/bull/g,p.bullet)(),p.list=l(p.list)(/bull/g,p.bullet)("hr","\\n+(?=\\1?(?:[-*_] *){3,}(?:\\n+|$))")("def","\\n+(?="+p.def.source+")")(),p.blockquote=l(p.blockquote)("def",p.def)(),p._tag="(?!(?:a|em|strong|small|s|cite|q|dfn|abbr|data|time|code|var|samp|kbd|sub|sup|i|b|u|mark|ruby|rt|rp|bdi|bdo|span|br|wbr|ins|del|img)\\b)\\w+(?!:/|[^\\w\\s@]*@)\\b",p.html=l(p.html)("comment",//)("closed",/<(tag)[\s\S]+?<\/\1>/)("closing",/])*?>/)(/tag/g,p._tag)(),p.paragraph=l(p.paragraph)("hr",p.hr)("heading",p.heading)("lheading",p.lheading)("blockquote",p.blockquote)("tag","<"+p._tag)("def",p.def)(),p.normal=h({},p),p.gfm=h({},p.normal,{fences:/^ *(`{3,}|~{3,}) *(\S+)? *\n([\s\S]+?)\s*\1 *(?:\n+|$)/,paragraph:/^/}),p.gfm.paragraph=l(p.paragraph)("(?!","(?!"+p.gfm.fences.source.replace("\\1","\\2")+"|"+p.list.source.replace("\\1","\\3")+"|")(),p.tables=h({},p.gfm,{nptable:/^ *(\S.*\|.*)\n *([-:]+ *\|[-| :]*)\n((?:.*\|.*(?:\n|$))*)\n*/,table:/^ *\|(.+)\n *\|( *[-:]+[-| :]*)\n((?: *\|.*(?:\n|$))*)\n*/}),e.rules=p,e.lex=function(t,n){var r=new e(n);return r.lex(t)},e.prototype.lex=function(e){return e=e.replace(/\r\n|\r/g,"\n").replace(/\t/g," ").replace(/\u00a0/g," ").replace(/\u2424/g,"\n"),this.token(e,!0)},e.prototype.token=function(e,t,n){for(var r,s,i,l,o,h,a,u,c,e=e.replace(/^ +$/gm,"");e;)if((i=this.rules.newline.exec(e))&&(e=e.substring(i[0].length),i[0].length>1&&this.tokens.push({type:"space"})),i=this.rules.code.exec(e))e=e.substring(i[0].length),i=i[0].replace(/^ {4}/gm,""),this.tokens.push({type:"code",text:this.options.pedantic?i:i.replace(/\n+$/,"")});else if(i=this.rules.fences.exec(e))e=e.substring(i[0].length),this.tokens.push({type:"code",lang:i[2],text:i[3]});else if(i=this.rules.heading.exec(e))e=e.substring(i[0].length),this.tokens.push({type:"heading",depth:i[1].length,text:i[2]});else if(t&&(i=this.rules.nptable.exec(e))){for(e=e.substring(i[0].length),h={type:"table",header:i[1].replace(/^ *| *\| *$/g,"").split(/ *\| */),align:i[2].replace(/^ *|\| *$/g,"").split(/ *\| */),cells:i[3].replace(/\n$/,"").split("\n")},u=0;u ?/gm,""),this.token(i,t,!0),this.tokens.push({type:"blockquote_end"});else if(i=this.rules.list.exec(e)){for(e=e.substring(i[0].length),l=i[2],this.tokens.push({type:"list_start",ordered:l.length>1}),i=i[0].match(this.rules.item),r=!1,c=i.length,u=0;c>u;u++)h=i[u],a=h.length,h=h.replace(/^ *([*+-]|\d+\.) +/,""),~h.indexOf("\n ")&&(a-=h.length,h=this.options.pedantic?h.replace(/^ {1,4}/gm,""):h.replace(new RegExp("^ {1,"+a+"}","gm"),"")),this.options.smartLists&&u!==c-1&&(o=p.bullet.exec(i[u+1])[0],l===o||l.length>1&&o.length>1||(e=i.slice(u+1).join("\n")+e,u=c-1)),s=r||/\n\n(?!\s*$)/.test(h),u!==c-1&&(r="\n"===h.charAt(h.length-1),s||(s=r)),this.tokens.push({type:s?"loose_item_start":"list_item_start"}),this.token(h,!1,n),this.tokens.push({type:"list_item_end"});this.tokens.push({type:"list_end"})}else if(i=this.rules.html.exec(e))e=e.substring(i[0].length),this.tokens.push({type:this.options.sanitize?"paragraph":"html",pre:"pre"===i[1]||"script"===i[1]||"style"===i[1],text:i[0]});else if(!n&&t&&(i=this.rules.def.exec(e)))e=e.substring(i[0].length),this.tokens.links[i[1].toLowerCase()]={href:i[2],title:i[3]};else if(t&&(i=this.rules.table.exec(e))){for(e=e.substring(i[0].length),h={type:"table",header:i[1].replace(/^ *| *\| *$/g,"").split(/ *\| */),align:i[2].replace(/^ *|\| *$/g,"").split(/ *\| */),cells:i[3].replace(/(?: *\| *)?\n$/,"").split("\n")},u=0;u])/,autolink:/^<([^ >]+(@|:\/)[^ >]+)>/,url:o,tag:/^|^<\/?\w+(?:"[^"]*"|'[^']*'|[^'">])*?>/,link:/^!?\[(inside)\]\(href\)/,reflink:/^!?\[(inside)\]\s*\[([^\]]*)\]/,nolink:/^!?\[((?:\[[^\]]*\]|[^\[\]])*)\]/,strong:/^__([\s\S]+?)__(?!_)|^\*\*([\s\S]+?)\*\*(?!\*)/,em:/^\b_((?:__|[\s\S])+?)_\b|^\*((?:\*\*|[\s\S])+?)\*(?!\*)/,code:/^(`+)\s*([\s\S]*?[^`])\s*\1(?!`)/,br:/^ {2,}\n(?!\s*$)/,del:o,text:/^[\s\S]+?(?=[\\?(?:\s+['"]([\s\S]*?)['"])?\s*/,u.link=l(u.link)("inside",u._inside)("href",u._href)(),u.reflink=l(u.reflink)("inside",u._inside)(),u.normal=h({},u),u.pedantic=h({},u.normal,{strong:/^__(?=\S)([\s\S]*?\S)__(?!_)|^\*\*(?=\S)([\s\S]*?\S)\*\*(?!\*)/,em:/^_(?=\S)([\s\S]*?\S)_(?!_)|^\*(?=\S)([\s\S]*?\S)\*(?!\*)/}),u.gfm=h({},u.normal,{escape:l(u.escape)("])","~|])")(),url:/^(https?:\/\/[^\s<]+[^<.,:;"')\]\s])/,del:/^~~(?=\S)([\s\S]*?\S)~~/,text:l(u.text)("]|","~]|")("|","|https?://|")()}),u.breaks=h({},u.gfm,{br:l(u.br)("{2,}","*")(),text:l(u.gfm.text)("{2,}","*")()}),t.rules=u,t.output=function(e,n,r){var s=new t(n,r);return s.output(e)},t.prototype.output=function(e){for(var t,n,r,i,l="";e;)if(i=this.rules.escape.exec(e))e=e.substring(i[0].length),l+=i[1];else if(i=this.rules.autolink.exec(e))e=e.substring(i[0].length),"@"===i[2]?(n=this.mangle(":"===i[1].charAt(6)?i[1].substring(7):i[1]),r=this.mangle("mailto:")+n):(n=s(i[1]),r=n),l+=this.renderer.link(r,null,n);else if(this.inLink||!(i=this.rules.url.exec(e))){if(i=this.rules.tag.exec(e))!this.inLink&&/^
                /i.test(i[0])&&(this.inLink=!1),e=e.substring(i[0].length),l+=this.options.sanitize?s(i[0]):i[0];else if(i=this.rules.link.exec(e))e=e.substring(i[0].length),this.inLink=!0,l+=this.outputLink(i,{href:i[2],title:i[3]}),this.inLink=!1;else if((i=this.rules.reflink.exec(e))||(i=this.rules.nolink.exec(e))){if(e=e.substring(i[0].length),t=(i[2]||i[1]).replace(/\s+/g," "),t=this.links[t.toLowerCase()],!t||!t.href){l+=i[0].charAt(0),e=i[0].substring(1)+e;continue}this.inLink=!0,l+=this.outputLink(i,t),this.inLink=!1}else if(i=this.rules.strong.exec(e))e=e.substring(i[0].length),l+=this.renderer.strong(this.output(i[2]||i[1]));else if(i=this.rules.em.exec(e))e=e.substring(i[0].length),l+=this.renderer.em(this.output(i[2]||i[1]));else if(i=this.rules.code.exec(e))e=e.substring(i[0].length),l+=this.renderer.codespan(s(i[2],!0));else if(i=this.rules.br.exec(e))e=e.substring(i[0].length),l+=this.renderer.br();else if(i=this.rules.del.exec(e))e=e.substring(i[0].length),l+=this.renderer.del(this.output(i[1]));else if(i=this.rules.text.exec(e))e=e.substring(i[0].length),l+=s(this.smartypants(i[0]));else if(e)throw new Error("Infinite loop on byte: "+e.charCodeAt(0))}else e=e.substring(i[0].length),n=s(i[1]),r=n,l+=this.renderer.link(r,null,n);return l},t.prototype.outputLink=function(e,t){var n=s(t.href),r=t.title?s(t.title):null;return"!"!==e[0].charAt(0)?this.renderer.link(n,r,this.output(e[1])):this.renderer.image(n,r,s(e[1]))},t.prototype.smartypants=function(e){return this.options.smartypants?e.replace(/--/g,"—").replace(/(^|[-\u2014/(\[{"\s])'/g,"$1‘").replace(/'/g,"’").replace(/(^|[-\u2014/(\[{\u2018\s])"/g,"$1“").replace(/"/g,"”").replace(/\.{3}/g,"…"):e},t.prototype.mangle=function(e){for(var t,n="",r=e.length,s=0;r>s;s++)t=e.charCodeAt(s),Math.random()>.5&&(t="x"+t.toString(16)),n+="&#"+t+";";return n},n.prototype.code=function(e,t,n){if(this.options.highlight){var r=this.options.highlight(e,t);null!=r&&r!==e&&(n=!0,e=r)}return t?'
                '+(n?e:s(e,!0))+"\n
                \n":"
                "+(n?e:s(e,!0))+"\n
                "},n.prototype.blockquote=function(e){return"
                \n"+e+"
                \n"},n.prototype.html=function(e){return e},n.prototype.heading=function(e,t,n){return"'+e+"\n"},n.prototype.hr=function(){return this.options.xhtml?"
                \n":"
                \n"},n.prototype.list=function(e,t){var n=t?"ol":"ul";return"<"+n+">\n"+e+"\n"},n.prototype.listitem=function(e){return"
              • "+e+"
              • \n"},n.prototype.paragraph=function(e){return"

                "+e+"

                \n"},n.prototype.table=function(e,t){return"\n\n"+e+"\n\n"+t+"\n
                \n"},n.prototype.tablerow=function(e){return"\n"+e+"\n"},n.prototype.tablecell=function(e,t){var n=t.header?"th":"td",r=t.align?"<"+n+' style="text-align:'+t.align+'">':"<"+n+">";return r+e+"\n"},n.prototype.strong=function(e){return""+e+""},n.prototype.em=function(e){return""+e+""},n.prototype.codespan=function(e){return""+e+""},n.prototype.br=function(){return this.options.xhtml?"
                ":"
                "},n.prototype.del=function(e){return""+e+""},n.prototype.link=function(e,t,n){if(this.options.sanitize){try{var r=decodeURIComponent(i(e)).replace(/[^\w:]/g,"").toLowerCase()}catch(s){return""}if(0===r.indexOf("javascript:")||0===r.indexOf("vbscript:"))return""}var l='
                "},n.prototype.image=function(e,t,n){var r=''+n+'":">"},r.parse=function(e,t,n){var s=new r(t,n);return s.parse(e)},r.prototype.parse=function(e){this.inline=new t(e.links,this.options,this.renderer),this.tokens=e.reverse();for(var n="";this.next();)n+=this.tok();return n},r.prototype.next=function(){return this.token=this.tokens.pop()},r.prototype.peek=function(){return this.tokens[this.tokens.length-1]||0},r.prototype.parseText=function(){for(var e=this.token.text;"text"===this.peek().type;)e+="\n"+this.next().text;return this.inline.output(e)},r.prototype.tok=function(){switch(this.token.type){case"space":return"";case"hr":return this.renderer.hr();case"heading":return this.renderer.heading(this.inline.output(this.token.text),this.token.depth,this.token.text);case"code":return this.renderer.code(this.token.text,this.token.lang,this.token.escaped);case"table":var e,t,n,r,s,i="",l="";for(n="",e=0;ebody{font-family: sans-serif;}

                reveal.js multiplex server.

                Generate token'); - res.end(); - }); - stream.on('readable', function() { - stream.pipe(res); - }); + fs.createReadStream(opts.baseDir + '/index.html').pipe(res); }); app.get("/token", function(req,res) { @@ -55,7 +47,7 @@ var createHash = function(secret) { }; // Actually listen -server.listen( opts.port || null ); +app.listen(opts.port || null); var brown = '\033[33m', green = '\033[32m', diff --git a/doc/pub/Splines/html/reveal.js/plugin/multiplex/master.js b/doc/pub/Splines/html/reveal.js/plugin/multiplex/master.js index 7f4bf4511..b6a7eb7dc 100644 --- a/doc/pub/Splines/html/reveal.js/plugin/multiplex/master.js +++ b/doc/pub/Splines/html/reveal.js/plugin/multiplex/master.js @@ -1,34 +1,51 @@ (function() { - // Don't emit events from inside of notes windows if ( window.location.search.match( /receiver/gi ) ) { return; } var multiplex = Reveal.getConfig().multiplex; - var socket = io.connect( multiplex.url ); + var socket = io.connect(multiplex.url); - function post() { + var notify = function( slideElement, indexh, indexv, origin ) { + if( typeof origin === 'undefined' && origin !== 'remote' ) { + var nextindexh; + var nextindexv; - var messageData = { - state: Reveal.getState(), - secret: multiplex.secret, - socketId: multiplex.id - }; + var fragmentindex = Reveal.getIndices().f; + if (typeof fragmentindex == 'undefined') { + fragmentindex = 0; + } - socket.emit( 'multiplex-statechanged', messageData ); + if (slideElement.nextElementSibling && slideElement.parentNode.nodeName == 'SECTION') { + nextindexh = indexh; + nextindexv = indexv + 1; + } else { + nextindexh = indexh + 1; + nextindexv = 0; + } + var slideData = { + indexh : indexh, + indexv : indexv, + indexf : fragmentindex, + nextindexh : nextindexh, + nextindexv : nextindexv, + secret: multiplex.secret, + socketId : multiplex.id + }; + + socket.emit('slidechanged', slideData); + } + } + + Reveal.addEventListener( 'slidechanged', function( event ) { + notify( event.currentSlide, event.indexh, event.indexv, event.origin ); + } ); + + var fragmentNotify = function( event ) { + notify( Reveal.getCurrentSlide(), Reveal.getIndices().h, Reveal.getIndices().v, event.origin ); }; - // post once the page is loaded, so the client follows also on "open URL". - window.addEventListener( 'load', post ); - - // Monitor events that trigger a change in state - Reveal.addEventListener( 'slidechanged', post ); - Reveal.addEventListener( 'fragmentshown', post ); - Reveal.addEventListener( 'fragmenthidden', post ); - Reveal.addEventListener( 'overviewhidden', post ); - Reveal.addEventListener( 'overviewshown', post ); - Reveal.addEventListener( 'paused', post ); - Reveal.addEventListener( 'resumed', post ); - -}()); + Reveal.addEventListener( 'fragmentshown', fragmentNotify ); + Reveal.addEventListener( 'fragmenthidden', fragmentNotify ); +}()); \ No newline at end of file diff --git a/doc/pub/Splines/html/reveal.js/plugin/notes-server/client.js b/doc/pub/Splines/html/reveal.js/plugin/notes-server/client.js index 00b277baf..628586ffb 100644 --- a/doc/pub/Splines/html/reveal.js/plugin/notes-server/client.js +++ b/doc/pub/Splines/html/reveal.js/plugin/notes-server/client.js @@ -41,15 +41,10 @@ } // When a new notes window connects, post our current state - socket.on( 'new-subscriber', function( data ) { + socket.on( 'connect', function( data ) { post(); } ); - // When the state changes from inside of the speaker view - socket.on( 'statechanged-speaker', function( data ) { - Reveal.setState( data.state ); - } ); - // Monitor events that trigger a change in state Reveal.addEventListener( 'slidechanged', post ); Reveal.addEventListener( 'fragmentshown', post ); diff --git a/doc/pub/Splines/html/reveal.js/plugin/notes-server/index.js b/doc/pub/Splines/html/reveal.js/plugin/notes-server/index.js index b95f07188..df917f112 100644 --- a/doc/pub/Splines/html/reveal.js/plugin/notes-server/index.js +++ b/doc/pub/Splines/html/reveal.js/plugin/notes-server/index.js @@ -1,40 +1,37 @@ -var http = require('http'); var express = require('express'); var fs = require('fs'); var io = require('socket.io'); +var _ = require('underscore'); var Mustache = require('mustache'); -var app = express(); +var app = express.createServer(); var staticDir = express.static; -var server = http.createServer(app); -io = io(server); +io = io.listen(app); var opts = { port : 1947, baseDir : __dirname + '/../../' }; -io.on( 'connection', function( socket ) { +io.sockets.on( 'connection', function( socket ) { - socket.on( 'new-subscriber', function( data ) { - socket.broadcast.emit( 'new-subscriber', data ); + socket.on( 'connect', function( data ) { + socket.broadcast.emit( 'connect', data ); }); socket.on( 'statechanged', function( data ) { - delete data.state.overview; socket.broadcast.emit( 'statechanged', data ); }); - socket.on( 'statechanged-speaker', function( data ) { - delete data.state.overview; - socket.broadcast.emit( 'statechanged-speaker', data ); - }); - }); -[ 'css', 'js', 'images', 'plugin', 'lib' ].forEach( function( dir ) { - app.use( '/' + dir, staticDir( opts.baseDir + dir ) ); +app.configure( function() { + + [ 'css', 'js', 'images', 'plugin', 'lib' ].forEach( function( dir ) { + app.use( '/' + dir, staticDir( opts.baseDir + dir ) ); + }); + }); app.get('/', function( req, res ) { @@ -55,7 +52,7 @@ app.get( '/notes/:socketId', function( req, res ) { }); // Actually listen -server.listen( opts.port || null ); +app.listen( opts.port || null ); var brown = '\033[33m', green = '\033[32m', @@ -65,5 +62,5 @@ var slidesLocation = 'http://localhost' + ( opts.port ? ( ':' + opts.port ) : '' console.log( brown + 'reveal.js - Speaker Notes' + reset ); console.log( '1. Open the slides at ' + green + slidesLocation + reset ); -console.log( '2. Click on the link in your JS console to go to the notes page' ); +console.log( '2. Click on the link your JS console to go to the notes page' ); console.log( '3. Advance through your slides and your notes will advance automatically' ); diff --git a/doc/pub/Splines/html/reveal.js/plugin/notes-server/notes.html b/doc/pub/Splines/html/reveal.js/plugin/notes-server/notes.html index ab8c5b17a..72d0317f1 100644 --- a/doc/pub/Splines/html/reveal.js/plugin/notes-server/notes.html +++ b/doc/pub/Splines/html/reveal.js/plugin/notes-server/notes.html @@ -8,7 +8,6 @@ @@ -247,7 +152,7 @@
                -
                Upcoming
                +
                UPCOMING:

                Time Click to Reset

                @@ -265,10 +170,6 @@
                -
                - - -
                @@ -281,20 +182,11 @@ currentState, currentSlide, upcomingSlide, - layoutLabel, - layoutDropdown, connected = false; var socket = io.connect( window.location.origin ), socketId = '{{socketId}}'; - var SPEAKER_LAYOUTS = { - 'default': 'Default', - 'wide': 'Wide', - 'tall': 'Tall', - 'notes-only': 'Notes only' - }; - socket.on( 'statechanged', function( data ) { // ignore data from sockets that aren't ours @@ -303,6 +195,7 @@ if( connected === false ) { connected = true; + setupIframes( data ); setupKeyboard(); setupNotes(); setupTimer(); @@ -313,28 +206,13 @@ } ); - setupLayout(); - - // Load our presentation iframes - setupIframes(); - - // Once the iframes have loaded, emit a signal saying there's - // a new subscriber which will trigger a 'statechanged' - // message to be sent back window.addEventListener( 'message', function( event ) { var data = JSON.parse( event.data ); if( data && data.namespace === 'reveal' ) { if( /ready/.test( data.eventName ) ) { - socket.emit( 'new-subscriber', { socketId: socketId } ); - } - } - - // Messages sent by reveal.js inside of the current slide preview - if( data && data.namespace === 'reveal' ) { - if( /slidechanged|fragmentshown|fragmenthidden|overviewshown|overviewhidden|paused|resumed/.test( data.eventName ) && currentState !== JSON.stringify( data.state ) ) { - socket.emit( 'statechanged-speaker', { state: data.state } ); + socket.emit( 'connect', { socketId: socketId } ); } } @@ -389,7 +267,7 @@ /** * Creates the preview iframes. */ - function setupIframes() { + function setupIframes( data ) { var params = [ 'receiver', @@ -399,8 +277,9 @@ 'backgroundTransition=none' ].join( '&' ); - var currentURL = '/?' + params + '&postMessageEvents=true'; - var upcomingURL = '/?' + params + '&controls=false'; + var hash = '#/' + data.state.indexh + '/' + data.state.indexv; + var currentURL = '/?' + params + '&postMessageEvents=true' + hash; + var upcomingURL = '/?' + params + '&controls=false' + hash; currentSlide = document.createElement( 'iframe' ); currentSlide.setAttribute( 'width', 1280 ); @@ -472,74 +351,6 @@ } - /** - * Sets up the speaker view layout and layout selector. - */ - function setupLayout() { - - layoutDropdown = document.querySelector( '.speaker-layout-dropdown' ); - layoutLabel = document.querySelector( '.speaker-layout-label' ); - - // Render the list of available layouts - for( var id in SPEAKER_LAYOUTS ) { - var option = document.createElement( 'option' ); - option.setAttribute( 'value', id ); - option.textContent = SPEAKER_LAYOUTS[ id ]; - layoutDropdown.appendChild( option ); - } - - // Monitor the dropdown for changes - layoutDropdown.addEventListener( 'change', function( event ) { - - setLayout( layoutDropdown.value ); - - }, false ); - - // Restore any currently persisted layout - setLayout( getLayout() ); - - } - - /** - * Sets a new speaker view layout. The layout is persisted - * in local storage. - */ - function setLayout( value ) { - - var title = SPEAKER_LAYOUTS[ value ]; - - layoutLabel.innerHTML = 'Layout' + ( title ? ( ': ' + title ) : '' ); - layoutDropdown.value = value; - - document.body.setAttribute( 'data-speaker-layout', value ); - - // Persist locally - if( window.localStorage ) { - window.localStorage.setItem( 'reveal-speaker-layout', value ); - } - - } - - /** - * Returns the ID of the most recently set speaker layout - * or our default layout if none has been set. - */ - function getLayout() { - - if( window.localStorage ) { - var layout = window.localStorage.getItem( 'reveal-speaker-layout' ); - if( layout ) { - return layout; - } - } - - // Default to the first record in the layouts hash - for( var id in SPEAKER_LAYOUTS ) { - return id; - } - - } - function zeroPadInteger( num ) { var str = '00' + parseInt( num ); diff --git a/doc/pub/Splines/html/reveal.js/plugin/notes/notes.html b/doc/pub/Splines/html/reveal.js/plugin/notes/notes.html index 4c5b799b5..0cc8cf612 100644 --- a/doc/pub/Splines/html/reveal.js/plugin/notes/notes.html +++ b/doc/pub/Splines/html/reveal.js/plugin/notes/notes.html @@ -8,7 +8,6 @@