From 759aa14da5b09b1777f48b78a4a882aaee3d1a88 Mon Sep 17 00:00:00 2001 From: Morten Hjorth-Jensen Date: Sun, 28 Mar 2021 22:16:58 -0400 Subject: [PATCH] test --- doc/BookChapters/chapter1.dlog | 17 - doc/BookChapters/chapter2.dlog | 7 - doc/BookChapters/chapter3.dlog | 40 - doc/BookChapters/chapter4.dlog | 7 - doc/LectureNotes/_add.yml | 25 + .../_build/.doctrees/chapter1.doctree | Bin 0 -> 432356 bytes .../_build/.doctrees/chapter2.doctree | Bin 0 -> 169250 bytes .../_build/.doctrees/chapter3.doctree | Bin 0 -> 144166 bytes .../_build/.doctrees/chapter4.doctree | Bin 0 -> 281174 bytes .../_build/.doctrees/content.doctree | Bin 0 -> 2850 bytes .../_build/.doctrees/environment.pickle | Bin 0 -> 74321 bytes .../_build/.doctrees/glue_cache.json | 1 + .../_build/.doctrees/intro.doctree | Bin 0 -> 47257 bytes .../_build/.doctrees/schedule.doctree | Bin 0 -> 6674 bytes .../_build/.doctrees/teachers.doctree | Bin 0 -> 9008 bytes .../_build/.doctrees/textbooks.doctree | Bin 0 -> 23977 bytes doc/LectureNotes/_build/html/.buildinfo | 4 + .../_build/html/_images/chapter2_25_2.png | Bin 0 -> 4742 bytes ...-main.c949a650a448cc0ae9fd3441c0e17fb0.css | 1 + ...ables.06eb56fa6e07937060861dad626602ad.css | 7 + .../_build/html/_sources/chapter1.ipynb | 4066 ++++++ .../_build/html/_sources/chapter2.ipynb | 1355 ++ .../_build/html/_sources}/chapter3.ipynb | 0 .../_build/html/_sources/chapter4.ipynb | 2875 ++++ .../_build/html/_sources/content.md | 5 + .../_build/html/_sources/intro.md | 145 + .../_build/html/_sources/schedule.md | 14 + .../_build/html/_sources/teachers.md | 23 + .../_build/html/_sources/textbooks.md | 38 + .../_build/html/_static/__init__.py | 0 .../__pycache__/__init__.cpython-38.pyc | Bin 0 -> 185 bytes .../_build/html/_static/basic.css | 856 ++ .../_build/html/_static/clipboard.min.js | 7 + .../_build/html/_static/copy-button.svg | 5 + .../_build/html/_static/copybutton.css | 67 + .../_build/html/_static/copybutton.js | 153 + .../_build/html/_static/copybutton_funcs.js | 47 + ...index.f658d18f9b420779cfdf24aa0a7e2d77.css | 6 + .../_build/html/_static/doctools.js | 316 + .../html/_static/documentation_options.js | 12 + doc/LectureNotes/_build/html/_static/file.png | Bin 0 -> 286 bytes .../html/_static/images/logo_binder.svg | 19 + .../_build/html/_static/images/logo_colab.png | Bin 0 -> 7601 bytes .../html/_static/images/logo_jupyterhub.svg | 1 + .../_build/html/_static/jquery-3.5.1.js | 10872 ++++++++++++++++ .../_build/html/_static/jquery.js | 2 + .../_static/js/index.d3f166471bb80abb5163.js | 32 + .../_build/html/_static/language_data.js | 297 + doc/LectureNotes/_build/html/_static/logo.png | Bin 0 -> 700391 bytes .../_build/html/_static/minus.png | Bin 0 -> 90 bytes .../_build/html/_static/mystnb.css | 183 + ...-main.c949a650a448cc0ae9fd3441c0e17fb0.css | 1 + ...ables.06eb56fa6e07937060861dad626602ad.css | 7 + doc/LectureNotes/_build/html/_static/plus.png | Bin 0 -> 90 bytes .../_build/html/_static/pygments.css | 74 + .../_build/html/_static/searchtools.js | 514 + ...-theme.7d483ff0a819d6edff12ce0b1ead3928.js | 36 + .../_build/html/_static/sphinx-book-theme.css | 1 + ...theme.e7340bb3dbd8dde6db86f25597f54a1b.css | 5 + .../_build/html/_static/sphinx-thebe.css | 120 + .../_build/html/_static/sphinx-thebe.js | 96 + .../_build/html/_static/togglebutton.css | 90 + .../_build/html/_static/togglebutton.js | 76 + .../_build/html/_static/underscore-1.3.1.js | 999 ++ .../_build/html/_static/underscore.js | 31 + .../vendor/fontawesome/5.13.0/LICENSE.txt | 34 + .../vendor/fontawesome/5.13.0/css/all.min.css | 5 + .../5.13.0/webfonts/fa-brands-400.eot | Bin 0 -> 133034 bytes .../5.13.0/webfonts/fa-brands-400.svg | 3570 +++++ .../5.13.0/webfonts/fa-brands-400.ttf | Bin 0 -> 132728 bytes .../5.13.0/webfonts/fa-brands-400.woff | Bin 0 -> 89824 bytes .../5.13.0/webfonts/fa-brands-400.woff2 | Bin 0 -> 76612 bytes .../5.13.0/webfonts/fa-regular-400.eot | Bin 0 -> 34390 bytes .../5.13.0/webfonts/fa-regular-400.svg | 803 ++ .../5.13.0/webfonts/fa-regular-400.ttf | Bin 0 -> 34092 bytes .../5.13.0/webfonts/fa-regular-400.woff | Bin 0 -> 16800 bytes .../5.13.0/webfonts/fa-regular-400.woff2 | Bin 0 -> 13584 bytes .../5.13.0/webfonts/fa-solid-900.eot | Bin 0 -> 202902 bytes .../5.13.0/webfonts/fa-solid-900.svg | 4938 +++++++ .../5.13.0/webfonts/fa-solid-900.ttf | Bin 0 -> 202616 bytes .../5.13.0/webfonts/fa-solid-900.woff | Bin 0 -> 103300 bytes .../5.13.0/webfonts/fa-solid-900.woff2 | Bin 0 -> 79444 bytes .../vendor/lato_latin-ext/1.44.1/LICENSE.md | 20 + .../files/lato-latin-ext-100-italic.woff | Bin 0 -> 23416 bytes .../files/lato-latin-ext-100-italic.woff2 | Bin 0 -> 18228 bytes .../1.44.1/files/lato-latin-ext-100.woff | Bin 0 -> 29264 bytes .../1.44.1/files/lato-latin-ext-100.woff2 | Bin 0 -> 23300 bytes .../files/lato-latin-ext-300-italic.woff | Bin 0 -> 24056 bytes .../files/lato-latin-ext-300-italic.woff2 | Bin 0 -> 18868 bytes .../1.44.1/files/lato-latin-ext-300.woff | Bin 0 -> 32196 bytes .../1.44.1/files/lato-latin-ext-300.woff2 | Bin 0 -> 24836 bytes .../files/lato-latin-ext-400-italic.woff | Bin 0 -> 32220 bytes .../files/lato-latin-ext-400-italic.woff2 | Bin 0 -> 26312 bytes .../1.44.1/files/lato-latin-ext-400.woff | Bin 0 -> 30924 bytes .../1.44.1/files/lato-latin-ext-400.woff2 | Bin 0 -> 25320 bytes .../files/lato-latin-ext-700-italic.woff | Bin 0 -> 32564 bytes .../files/lato-latin-ext-700-italic.woff2 | Bin 0 -> 26344 bytes .../1.44.1/files/lato-latin-ext-700.woff | Bin 0 -> 30356 bytes .../1.44.1/files/lato-latin-ext-700.woff2 | Bin 0 -> 24712 bytes .../files/lato-latin-ext-900-italic.woff | Bin 0 -> 31260 bytes .../files/lato-latin-ext-900-italic.woff2 | Bin 0 -> 25636 bytes .../1.44.1/files/lato-latin-ext-900.woff | Bin 0 -> 29700 bytes .../1.44.1/files/lato-latin-ext-900.woff2 | Bin 0 -> 24344 bytes .../vendor/lato_latin-ext/1.44.1/index.css | 120 + .../vendor/open-sans_all/1.44.1/LICENSE.md | 20 + .../files/open-sans-all-400-italic.woff | Bin 0 -> 53024 bytes .../files/open-sans-all-400-italic.woff2 | Bin 0 -> 41076 bytes .../1.44.1/files/open-sans-all-400.woff | Bin 0 -> 55268 bytes .../1.44.1/files/open-sans-all-400.woff2 | Bin 0 -> 43236 bytes .../vendor/open-sans_all/1.44.1/index.css | 120 + .../_build/html/_static/webpack-macros.html | 28 + doc/LectureNotes/_build/html/chapter1.html | 2890 ++++ doc/LectureNotes/_build/html/chapter2.html | 1371 ++ doc/LectureNotes/_build/html/chapter3.html | 1102 ++ doc/LectureNotes/_build/html/chapter4.html | 2217 ++++ doc/LectureNotes/_build/html/content.html | 275 + doc/LectureNotes/_build/html/genindex.html | 240 + doc/LectureNotes/_build/html/index.html | 2 + doc/LectureNotes/_build/html/intro.html | 484 + doc/LectureNotes/_build/html/objects.inv | 7 + .../_build/html/reports/chapter1.log | 31 + .../_build/html/reports/chapter2.log | 104 + .../_build/html/reports/chapter4.log | 89 + doc/LectureNotes/_build/html/schedule.html | 288 + doc/LectureNotes/_build/html/search.html | 259 + doc/LectureNotes/_build/html/searchindex.js | 1 + doc/LectureNotes/_build/html/teachers.html | 320 + doc/LectureNotes/_build/html/textbooks.html | 328 + .../_build/jupyter_execute/chapter1.ipynb | 4129 ++++++ .../_build/jupyter_execute/chapter1.py | 2454 ++++ .../_build/jupyter_execute/chapter2.ipynb | 1420 ++ .../_build/jupyter_execute/chapter2.py | 1042 ++ .../_build/jupyter_execute/chapter2_25_2.png | Bin 0 -> 4742 bytes .../_build/jupyter_execute/chapter3.ipynb | 1508 +++ .../_build/jupyter_execute/chapter3.py | 790 ++ .../_build/jupyter_execute/chapter4.ipynb | 2900 +++++ .../_build/jupyter_execute/chapter4.py | 1752 +++ doc/LectureNotes/_toc.yml | 12 - doc/LectureNotes/chapter3.ipynb | 1378 ++ doc/LectureNotes/chapter4.ipynb | 2875 ++++ 140 files changed, 63398 insertions(+), 83 deletions(-) delete mode 100644 doc/BookChapters/chapter1.dlog delete mode 100644 doc/BookChapters/chapter2.dlog delete mode 100644 doc/BookChapters/chapter3.dlog delete mode 100644 doc/BookChapters/chapter4.dlog create mode 100644 doc/LectureNotes/_add.yml create mode 100644 doc/LectureNotes/_build/.doctrees/chapter1.doctree create mode 100644 doc/LectureNotes/_build/.doctrees/chapter2.doctree create mode 100644 doc/LectureNotes/_build/.doctrees/chapter3.doctree create mode 100644 doc/LectureNotes/_build/.doctrees/chapter4.doctree create mode 100644 doc/LectureNotes/_build/.doctrees/content.doctree create mode 100644 doc/LectureNotes/_build/.doctrees/environment.pickle create mode 100644 doc/LectureNotes/_build/.doctrees/glue_cache.json create mode 100644 doc/LectureNotes/_build/.doctrees/intro.doctree create mode 100644 doc/LectureNotes/_build/.doctrees/schedule.doctree create mode 100644 doc/LectureNotes/_build/.doctrees/teachers.doctree create mode 100644 doc/LectureNotes/_build/.doctrees/textbooks.doctree create mode 100644 doc/LectureNotes/_build/html/.buildinfo create mode 100644 doc/LectureNotes/_build/html/_images/chapter2_25_2.png create mode 100644 doc/LectureNotes/_build/html/_panels_static/panels-main.c949a650a448cc0ae9fd3441c0e17fb0.css create mode 100644 doc/LectureNotes/_build/html/_panels_static/panels-variables.06eb56fa6e07937060861dad626602ad.css create mode 100644 doc/LectureNotes/_build/html/_sources/chapter1.ipynb create mode 100644 doc/LectureNotes/_build/html/_sources/chapter2.ipynb rename doc/{BookChapters => LectureNotes/_build/html/_sources}/chapter3.ipynb (100%) create mode 100644 doc/LectureNotes/_build/html/_sources/chapter4.ipynb create mode 100644 doc/LectureNotes/_build/html/_sources/content.md create mode 100644 doc/LectureNotes/_build/html/_sources/intro.md create mode 100644 doc/LectureNotes/_build/html/_sources/schedule.md create mode 100644 doc/LectureNotes/_build/html/_sources/teachers.md create mode 100644 doc/LectureNotes/_build/html/_sources/textbooks.md create mode 100644 doc/LectureNotes/_build/html/_static/__init__.py create mode 100644 doc/LectureNotes/_build/html/_static/__pycache__/__init__.cpython-38.pyc create mode 100644 doc/LectureNotes/_build/html/_static/basic.css create mode 100644 doc/LectureNotes/_build/html/_static/clipboard.min.js create mode 100644 doc/LectureNotes/_build/html/_static/copy-button.svg create mode 100644 doc/LectureNotes/_build/html/_static/copybutton.css create mode 100644 doc/LectureNotes/_build/html/_static/copybutton.js create mode 100644 doc/LectureNotes/_build/html/_static/copybutton_funcs.js create mode 100644 doc/LectureNotes/_build/html/_static/css/index.f658d18f9b420779cfdf24aa0a7e2d77.css create mode 100644 doc/LectureNotes/_build/html/_static/doctools.js create mode 100644 doc/LectureNotes/_build/html/_static/documentation_options.js create mode 100644 doc/LectureNotes/_build/html/_static/file.png create mode 100644 doc/LectureNotes/_build/html/_static/images/logo_binder.svg create mode 100644 doc/LectureNotes/_build/html/_static/images/logo_colab.png create mode 100644 doc/LectureNotes/_build/html/_static/images/logo_jupyterhub.svg create mode 100644 doc/LectureNotes/_build/html/_static/jquery-3.5.1.js create mode 100644 doc/LectureNotes/_build/html/_static/jquery.js create mode 100644 doc/LectureNotes/_build/html/_static/js/index.d3f166471bb80abb5163.js create mode 100644 doc/LectureNotes/_build/html/_static/language_data.js create mode 100644 doc/LectureNotes/_build/html/_static/logo.png create mode 100644 doc/LectureNotes/_build/html/_static/minus.png create mode 100644 doc/LectureNotes/_build/html/_static/mystnb.css create mode 100644 doc/LectureNotes/_build/html/_static/panels-main.c949a650a448cc0ae9fd3441c0e17fb0.css create mode 100644 doc/LectureNotes/_build/html/_static/panels-variables.06eb56fa6e07937060861dad626602ad.css create mode 100644 doc/LectureNotes/_build/html/_static/plus.png create mode 100644 doc/LectureNotes/_build/html/_static/pygments.css create mode 100644 doc/LectureNotes/_build/html/_static/searchtools.js create mode 100644 doc/LectureNotes/_build/html/_static/sphinx-book-theme.7d483ff0a819d6edff12ce0b1ead3928.js create mode 100644 doc/LectureNotes/_build/html/_static/sphinx-book-theme.css create mode 100644 doc/LectureNotes/_build/html/_static/sphinx-book-theme.e7340bb3dbd8dde6db86f25597f54a1b.css create mode 100644 doc/LectureNotes/_build/html/_static/sphinx-thebe.css create mode 100644 doc/LectureNotes/_build/html/_static/sphinx-thebe.js create mode 100644 doc/LectureNotes/_build/html/_static/togglebutton.css create mode 100644 doc/LectureNotes/_build/html/_static/togglebutton.js create mode 100644 doc/LectureNotes/_build/html/_static/underscore-1.3.1.js create mode 100644 doc/LectureNotes/_build/html/_static/underscore.js create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/LICENSE.txt create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/css/all.min.css create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-brands-400.eot create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-brands-400.svg create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-brands-400.ttf create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-brands-400.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-brands-400.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-regular-400.eot create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-regular-400.svg create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-regular-400.ttf create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-regular-400.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-regular-400.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-solid-900.eot create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-solid-900.svg create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-solid-900.ttf create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-solid-900.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/fontawesome/5.13.0/webfonts/fa-solid-900.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/LICENSE.md create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-100-italic.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-100-italic.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-100.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-100.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-300-italic.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-300-italic.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-300.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-300.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-400-italic.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-400-italic.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-400.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-400.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-700-italic.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-700-italic.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-700.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-700.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-900-italic.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-900-italic.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-900.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/files/lato-latin-ext-900.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/lato_latin-ext/1.44.1/index.css create mode 100644 doc/LectureNotes/_build/html/_static/vendor/open-sans_all/1.44.1/LICENSE.md create mode 100644 doc/LectureNotes/_build/html/_static/vendor/open-sans_all/1.44.1/files/open-sans-all-400-italic.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/open-sans_all/1.44.1/files/open-sans-all-400-italic.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/open-sans_all/1.44.1/files/open-sans-all-400.woff create mode 100644 doc/LectureNotes/_build/html/_static/vendor/open-sans_all/1.44.1/files/open-sans-all-400.woff2 create mode 100644 doc/LectureNotes/_build/html/_static/vendor/open-sans_all/1.44.1/index.css create mode 100644 doc/LectureNotes/_build/html/_static/webpack-macros.html create mode 100644 doc/LectureNotes/_build/html/chapter1.html create mode 100644 doc/LectureNotes/_build/html/chapter2.html create mode 100644 doc/LectureNotes/_build/html/chapter3.html create mode 100644 doc/LectureNotes/_build/html/chapter4.html create mode 100644 doc/LectureNotes/_build/html/content.html create mode 100644 doc/LectureNotes/_build/html/genindex.html create mode 100644 doc/LectureNotes/_build/html/index.html create mode 100644 doc/LectureNotes/_build/html/intro.html create mode 100644 doc/LectureNotes/_build/html/objects.inv create mode 100644 doc/LectureNotes/_build/html/reports/chapter1.log create mode 100644 doc/LectureNotes/_build/html/reports/chapter2.log create mode 100644 doc/LectureNotes/_build/html/reports/chapter4.log create mode 100644 doc/LectureNotes/_build/html/schedule.html create mode 100644 doc/LectureNotes/_build/html/search.html create mode 100644 doc/LectureNotes/_build/html/searchindex.js create mode 100644 doc/LectureNotes/_build/html/teachers.html create mode 100644 doc/LectureNotes/_build/html/textbooks.html create mode 100644 doc/LectureNotes/_build/jupyter_execute/chapter1.ipynb create mode 100644 doc/LectureNotes/_build/jupyter_execute/chapter1.py create mode 100644 doc/LectureNotes/_build/jupyter_execute/chapter2.ipynb create mode 100644 doc/LectureNotes/_build/jupyter_execute/chapter2.py create mode 100644 doc/LectureNotes/_build/jupyter_execute/chapter2_25_2.png create mode 100644 doc/LectureNotes/_build/jupyter_execute/chapter3.ipynb create mode 100644 doc/LectureNotes/_build/jupyter_execute/chapter3.py create mode 100644 doc/LectureNotes/_build/jupyter_execute/chapter4.ipynb create mode 100644 doc/LectureNotes/_build/jupyter_execute/chapter4.py create mode 100644 doc/LectureNotes/chapter3.ipynb create mode 100644 doc/LectureNotes/chapter4.ipynb diff --git a/doc/BookChapters/chapter1.dlog b/doc/BookChapters/chapter1.dlog deleted file mode 100644 index ae2474b0b..000000000 --- a/doc/BookChapters/chapter1.dlog +++ /dev/null @@ -1,17 +0,0 @@ -Translating doconce text in chapter1.do.txt to ipynb -*** replacing \bm{...} by \boldsymbol{...} (\bm is not supported by MathJax) - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{cases} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. -Failed to remove ans_at_end environment -Failed to remove sol_at_end environment -output in chapter1.ipynb diff --git a/doc/BookChapters/chapter2.dlog b/doc/BookChapters/chapter2.dlog deleted file mode 100644 index 3d924b166..000000000 --- a/doc/BookChapters/chapter2.dlog +++ /dev/null @@ -1,7 +0,0 @@ -Translating doconce text in chapter2.do.txt to ipynb -*** replacing \bm{...} by \boldsymbol{...} (\bm is not supported by MathJax) - -*** warning: latex envir \begin{eqnarray*} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. -Failed to remove ans_at_end environment -Failed to remove sol_at_end environment -output in chapter2.ipynb diff --git a/doc/BookChapters/chapter3.dlog b/doc/BookChapters/chapter3.dlog deleted file mode 100644 index 1d754e8e2..000000000 --- a/doc/BookChapters/chapter3.dlog +++ /dev/null @@ -1,40 +0,0 @@ -*** error: file has a mako construction ${\bf X}' - but seemingly no definition in <%...%>' - (it is not a command-line given mako variable either). - However, if this is a variable in a Makefile or Bash script - run with --no_mako - and you cannot use mako and Makefile or Bash variables - in the same document! - -Translating doconce text in chapter3.do.txt to ipynb -*** replacing \bm{...} by \boldsymbol{...} (\bm is not supported by MathJax) - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. - -*** warning: latex envir \begin{bmatrix} does not work well in Markdown. Stick to \[ ... \], equation, equation*, align, or align* environments in math environments. -Failed to remove ans_at_end environment -Failed to remove sol_at_end environment -output in chapter3.ipynb diff --git a/doc/BookChapters/chapter4.dlog b/doc/BookChapters/chapter4.dlog deleted file mode 100644 index 88a8be5bd..000000000 --- a/doc/BookChapters/chapter4.dlog +++ /dev/null @@ -1,7 +0,0 @@ -*** error: file has a mako construction ${\bf \bm{J}' - but seemingly no definition in <%...%>' - (it is not a command-line given mako variable either). - However, if this is a variable in a Makefile or Bash script - run with --no_mako - and you cannot use mako and Makefile or Bash variables - in the same document! - diff --git a/doc/LectureNotes/_add.yml b/doc/LectureNotes/_add.yml new file mode 100644 index 000000000..b39daaede --- /dev/null +++ b/doc/LectureNotes/_add.yml @@ -0,0 +1,25 @@ +- file: intro +- part: About the course + chapters: + - file: schedule + - file: teachers + - file: textbooks +- part: From Regression to Support Vector Machines + numbered: true + chapters: + - file: chapter1.ipynb + - file: chapter2.ipynb + - file: chapter3.ipynb + - file: chapter4.ipynb +- part: Deep Learning + numbered: true + chapters: + - file: chapter5.ipynb + - file: chapter6.ipynb + - file: chapter7.ipynb + - file: chapter8.ipynb +- part: Trees and Ensemble Methods + numbered: true + chapters: + - file: chapter9.ipynb + - file: chapter10.ipynb diff --git a/doc/LectureNotes/_build/.doctrees/chapter1.doctree b/doc/LectureNotes/_build/.doctrees/chapter1.doctree new file mode 100644 index 0000000000000000000000000000000000000000..858844cf42a8739d248799df1be2d773465ee244 GIT binary patch literal 432356 zcmeFa3z!_&bsk8N1PC-KiKIwKy;DRu~yo(<7n;Jw5{DZ zj-A*^?2T=C|8wrMs;j%Jy9xlcqCT6LuDbWuIp>~xUiaK{Kd|-V@4jivP4vIso7&Ba z-&ilW<#Hn^d+p8M_Ht0_*1bk&^E)?Z|Muq9&6(bgmb=ytx~-D8*}DlXN)^9WZh4K( zi<`YSQuB^qYZqky+uB~K;|C4hy|T5kt#aGN&6&!M-qw!asd?6`-h(gr4bN>kFM6vj zuiZw+yi;`Be#tpj^9YRgCUr*NThDsy1XPkP-?GV&gkR^HUR zwFV?MPu{(|w@uOM*@gGCy;ggnUa5MGw%1rV>$#v?V|C#q|LeT~tQ+(j1YCFlH0!p! zQvtwQC{^5M$7?MX{N_fZxLKL1+|ez906Tz)yV`P_6`Oz`KL0=XWiN1o6-SP8ai-Gg zG}{Lk7S`6*3SB?I)GV|+-Lmhs7QA``|E;^7hTBXp|W?+ z!b;HcYF?W_YhR&}(OeF^++E$fSg1GmJ$L^w{iHW-5CtvnTJ_U74#q3cw2l_%M*EN8avnhv-qcU93d2l?bXq~VD;6EEm7S4?nx2)@fMO`o2!b~Og10D`p9kyB>yQ1!ySpvNt2ZlHp>3yA zX$9TYO3>}#Py7bM`(B$ku_pMr?Kn9+4;oIlQO0WMxQ#OS3avc{ELidIAHU?*9Jk#> z59*pL`zy>n9mlP$1}(o+skfbO+bcVMgFCMHUaibuDZ3rlaT{)JgP_6K7?jqShwqEU zRbVEYsqE}6j0yRsCffL2LF5ZIk7DiK0ugGjhTEPjww#5ha%vIi-}My#icVZeeBf2eHNp1^RD*wJwAjgW2)hLQI4f ze6AD#%_gKq+{(A5N6{mtMieW1HGCI?2TUzX!jLd()7y4|aq`yPdJ}q3)0x1N5~%2{ z`VCT$c?(yk4cgbdTr;R`G=jSC)+Ak9057OJjljnUu(P}}lb=u|*=H9rnk z?q#S{=lq$Hf7$QMv#}ubI@EewJY*qP2wJO9?lip%vqXuUA>9?X%j^JA@BBmR10c7(k7h^-{zq-nIlu}q|d1#?CNj)9A%yPhRZ%|cpP z2xT|sWH>#NOmJEdC+#sAJo|{o;|gdCy2<)a2J3QayJ!C;eNapoH6CJV1?GW=wi zzB7f!v@l9F%w&+?)u*p5cw%1mePu~IR(IHXL^%cFhsI_NKgX!QWov-JZbIcq?;Yd{ z4>2LM)B2#YN?CcZlW;4U7Z!%%cOOzQ=Dbuume4+w%1W_-NIB4C&_82a5R+I*76R$V znhb}|jWE3+C!^KD$?Z!E<20j82KSR`j9UxxbmL3}du5xJcN`H2+ps$jXoZ8= zShZs+5wS>7V?@G2f>nW zYXDcN1GkM{;vL+k&RPK9tA%JAd-E>b-JpT!0YKOg(FePCu|?4acu4J_&dxnz)%7|( zqv)Q$;+JwT)BTcg*gHW04;hZ!ng;~%Z@Ut#QDZ=bD90MyOGGe~3o9ICCs*@}EqJwX z0m%*PciCX30@wI}fpN~3fB*_~*gVtlo@HpeoVJq!Mf1{@nBOaCvY z<|PZC3MGd?gZpqPRS@4i&G&uA!ts@eFcugK}h8tTdi=|?JTzWg4PNN)0wvCIU>A8 zQVN@cm)nllDa{rf=OlzHqG*WclAJRV2%%dY^5oE<5b2PMexun%Y>oqUjSUit*ot+^ zejC8keremOxd^}IAwzRYmaz+Kcn{ZX*c~DiF5U{ z+`(+9pS$zsn{GOz4AypDNa1$Lr|b!q(|sl`GzxaxE3?W!}`*1Ql1W| z%-9hOfWPe1Hvu0sI#E6Eob^hTMo)2|l`mn=q0gUzu_mac_RMgyk735G*zL1JN)$#7}#Ii8JB zvrx)v#57nPPA_9EjItOl4K|BPz9kp;+N|hJ&^ZQHV^Jv!N~J`ql%vD8J<=Cs{3&D| zo#-G%rwu2zM6o=I2y%`NhX(Pej+8!jD?LZyOZvSklK2`38cGd~6=>&6#Hc%&s3C0$ z(yL8Y^DrP-P2+1&+t8(ON84Sbv2EnoEa0>=4st@FZWu7iehW~L0RpqvhPi5bC~t8PQr1O+|B}hG?B5<0q24$OD)!@mS(Zl;_e2*4(nYfv_pZ!D}0^GhqI- ziA==}0Z9|-#B2jml2|R+Kn5nqZYf|u={v}bX#^cia*iSr6gTzCFpk{1U-J=U?NE#t ziD7+!5ofrydeEl6KoOQI8IJ^j-}XB|ies}PSV2f~Q*XTly9b04z)%*D6ErDZtt$+o zd=5D~XpX!h4oq`m5m+hk8)WoSswix?8rfRXu+2;B62>F!KfghFZ+S8w>7&h-e+3xP zEB@Inf{hRfDRqJtkpT?kc9C|~VotBPU@uCvb89)ul?HS?YMrAY^7mRL5SeR54@pB6 za9SCRM~ne06Itu{NeiRm`K_Fs>L?cgxcE5*w+M4rnW4zl^BOrh14w%!8}%}#z2=o+ z8pkkWi?_Ntd3?Oz%87z87~b+$FtEU7EW1>m9m^AP&r`Gir78wkd2A88*SREER*^HM4qhpzvKo&|F4VVU}XVMyE7CPz2 zN`uMKEG$C{n{38QgVFJ9%v1}ftVT?ORemh1+4$e%^^1XUwCnx+P}lpD2&$@`369RQ zV+ltEu5XcDMKb(_l>k3UyFp}w!=EYDkoUu`AUr%cBXCtkTaHA-e;^-^57Ury3@5?6 z65z;7*#yLi!D){PyL z>=#=XZH%gDwln*r0{bTZRLZQ+*jz-e$im_pJRrIcor80bf3i*dj)#5phN+1+f`Tiko(giylV`*D3Ui zk`WseUFIYtN@AtmYzL(YTu5CEgAIL*1@Z2!L$^6lNZ6k(LD=i#<7*Wk3WE*;V)Glv zm{>C+UWgNpGS*B%L)R(dpeus5R0)DW7V$W2KXV+Ez{0Urf(ziTAwEjFtL*8_G2UzD zXg8gf1;99ySr-9U4U5Q{Ep|~r!{HN`$h^W%Gg4#i@F9O0c)l0eTS$xTqA|AYDLM(P z)+p+G+4GvDkC9$Wo5*els+Khp)DWKN@8Ks119)Az6NWfD1zyV-tQIsu2?UEPv?nb> z!=OXkU#$RT8pebhp-cJ5uGrIurjr@HPJ06(=Q;whHJtBsRggT_3h0P39j!!9`7u(E zs#{-7Swn1;?9nXP%#sY*jE5!lY63+Y)8=?K9B*Ni)re!`WGP(0ki`NFW0lE>X|Ost zxje8iNi#|s=#Qy&mP#a9h-5Ti8l0X^DIhIuQV*2|kK@@0Y73pLMofiOHhROxDx(q8 zVD-*4YRp)0sfLMxf3yi#9BRV-`mF<^3{+B~(?Z!Ta*oMl;*dk042^b^>>U_IWVIk^ z9F-p6nhWa&)e%q;g33N9GXV+k$S6ngB{G&^tihrpBdvwJ__9|*P=M0fku6Aj=PEW) za+|^&q}(_*Hh7nHC7=UJ!cdWhgeV|^AO$vO+0!Rmr9)YJK5c>X_A~Ddy{tk8RM`Z% zf0St`^9r$t48tx@6-qb-vEsqB1KJ34m>@~s1SgawOP{jh1i~22iO>vU04)k3w48>w z#s-uGn7)j6m_q`~)OAN(l#`*P z7osM0#@eOyS8yIs^Kgs~?pwEpi9o6c=mi4*L)Qwv{3ZJ5Yt0GU#1C_84rbf_4gJPu== zhxnMqMQbQAhE` zLt`F?pX2l25D}BvR5Yn6n^;L$E^4wRZ_x=qvcNuPOf7}G{m5W$KgEYXJ`Ubfc2U!a z4X3NPjK;b8QTnK@nk;c0t;=XBFx|>b`fbu|R55cQ!1ov2iJPaL zy(iNco=WGZSjeOuDGe6y9AKX(ahEI6SQ=c8X5rmi=wvfq8jR+~ z2sIe1V4Pkt(2b7gYCDb&koO(fI)eSkv$UHoD>WkoiXGqnEmj0yD4c=|7O<5~`@($0 z1fgB*!6OdV3362MiHf68B`A-!@;bE4tQwYvzFALooK;e~SujrQ{?c&4`W>EjgM28%#jZiM{PY z?hGGsqO(F>iX&1I83n5-XvT?#E?T2XmLeJmOcz5S4{xr8GzNtCN*w7jgrl%Iz-J8T zQirm?TIcv2qH`Q6#NJ1fPZt$8a-a@E(*b3~HQO7dia41rkEE!%;IR-VMyQmx^jZAR zg97s$k;_vokOH!Nq^{+m2snbLG&@ZMTH6&0#F^OHv9nn)leMlh8W2n80Jbhh9a4&E zbKgM5wxAr+DDKNUdfU}a2zC{^H^(rmawlTn88H(38lQeXg7jd2m`c-~pw_b|ze%~f ziX80jzID)=wdgRoO&l3wybtw7zVpK^p;Q>-%Jtznh{4e;6bXU8rN3mG*+%39bb><~koxZ{SgjC3>$nQnoc&3I`rdVEp^ zV_`A*7!yMtX6K#px=!))px1_>2eklvpUjS<^W*4Vi@xLao0z6p1;;W`_!j6H4VVU} zhmx2dh7}SukAY{jA+5DyoFU!&re&Omr~P~6Sdt~HtYBQO(q~PJ6AUons2vVAARmyrhx9$j~9qX)tgU-8|e_*APT3kN@ZX;V3?us~bFK(?9c{r;9d<4Of z4sGMJo8_V*sy@O+K?(+aVHJmPWKBu(`H=C#c|G8Jl*E;{;88w>Do!a!#P_M>#fAf5 zRw)Tc=i{NIFI`^8Dbr*Q&8tl+j12cNj2Dl#fi9ey7-K67H>9%j-0L*a4x?34`gO9H z00dZNs`~FE@@S}Y#JB5xRITK%$gU$w_*H+8+RdKS{eDOHtNuRzR2IEh^k!A}e%f4W zdC)m7!xOHv(rs{o8Z6e3DO>b9Ym_&KV+y=J_4xyK)RAkXu2s}8piuZMgthQV6?Iy; z4G>yKeMqeRy35&4q?0%bBJ>t2vUDkbNouPrZVef9Y7KrZV_@{f*YY$>Rkv&nM?6OJ z_s0}}e^vA8&`yfKb-$$i{i!20B)V}i#DmUJzKk)dt5%U52e#udS;@oAbQD{ZCQr+S z8bMKJUmW?N@EgVNDDq~m`;!BRE8IxCP&|?=c=B~~U~;O535zP$o7Rl{wE(K&*sld> zA$so{n>9BTYwna?6N+Bl&#>lDi>>r5Qn?V$E>U$1k$o)El{p$N z3%qOIMMgC0oa#JQB%7+T4H$yX#tDYEtQ**C@eIoHd&NifEgZeV709@x7ss<;#5Jho z6;&_gYTvZDsEGGW=qMWwh1-@Paz?SfXbn+h7fH23(jz68))XpV1srAf__ju67m5Mx z54t(6IIyn7Wex~ZgGyXLj3E{_B{UM<^h(-eVEEjb0q!;2XGlPx_W25`pf6MGS=pB$ z=86@Sgq3PArb4wimM0_Nbj2cWvgA#MH8N+{_-Gu?^vP13BKGa#`UrIgBg&Z}rvCSXMZJ7J6ycHEaS7V?C=v*Tw$()5gDIcQmph=sW?vM}R( zw3s$WGT>qhp-hGw8ylUNw6nLIdPP~-WH4A7T;7$@+O{ytY{(RNox2~N@9W}bpt#XC z!Rv$?A(#85-cSpHJ;Co%=18o0@HAMHq&y#93X2jb<6xdb7|=L#b0t*5oSQRRQ>awD z4N%rxluDPU;#uwCYzPjDAn1Y;lF)CfL4d;$7|zKT?DG>C4`w=rl?)Qla?q_W^Io~t z7H3ze3MHv$54HpD&Sgtp-O9vAN4QYt3dn%-`E0$+%_+rb)a6472x*C?M9m>~RwIh8 zI)gb={&??Jni#bKA%5ufq+U){n~^bw`%3cSukTfyqadK&)ySP?ebB|JVb#Afr*A_) zrsZK8MzW}x(SR}Z9*&WvikBu}YYT&ky2ru0qM75iCyOm-;i;VB5B*!S2;DnZ-Av3K zy=~-*x?s2~m9@$`)mrvKAXw!1;Ei9e7Ij|>(1a)@mO^+jZ&%_)O_+J0XRK$QVZ0wr zJav>dF>GP?b{L6fK4v|Pfqu01(K5tnAJeJ#R`PJ>{r0@vd74+LKAYxZl>68V!qSTF zPCboAky}#K9VC?mx1>CH24WvMLeLVB`nc~x>^6A{SHykwq^T)j>LS^V zls4a`wHCB4%bt|{-0C8EZtg{x2bAZ{x}Eg79BSc&1XK?x;$f`w{dyhY0J0Q{C}{&f z7Kr7Lg4cCxKA&vn?@=tbz<|={I5~}IK!@XeGEWrFvfA2_M<31SUPRa~4;WD9JMSDv z1@0EPN~x=Ejmw*^CFy z`*K7g#Hp{CHuGb#i3RwW9mj^y(kPEl%A6LslaCPtd02XHc~MCW{G}yAj~-7||662Q z5?NqBrlf=AS9}-Bw&^$Fg1T*%I25Zrkr_r7lVmt3CUNh!@Gf86qK;{E zEE5V_SY(feisVU{$ z>|2OutuHVO_p`?6^{t1u*7uIyn>^G;;ZqCc_;jV=$qZ*e3%v|6QgegK597aZd2zW+ z2WGQ_pV-Ug8%{3!9pSM!^;PySXVrprPF24c zfi2N8`K($d3Tj>^2PT|J3%!Ylh@r-CDqBXxTL*HIq1WIxF<&0NzBB>N`l?ni6Dr~? z(gTMp*?VXT)dF#ZHr1jg{-wsJ`d2jVQ`Ns1XZBeu{R=hqe=fTw)JnSFe@)m=Y9-yz zP|5`NdOt`@`~{>Xt_xWtBVv?tO#4k990~{x6#TL-72t~mEP26x!WvSRkw4Rv_4eLV z>^3^xP{(|ZY)y)rQOL}A4iJA3W)%BG#z_-ns8uT+4$HT=d>x^+7S(nAEST!v+eZ5* zF>UU!>1%->ejYP^SnMUR!*@pjOeFJuRx!Q;on_v`FuW6RfL?gLgdJkg94%Eeh-R&f zEmaP*yJaLjH9PZk@w?(>DP0^+(2|BbEurQ{r4uyv77olRMKf`KL!ya?0MXtx&eXC% z`*}65M`hQ90!{Z@PS{ThG~Lfkk|of#Q)C_gFKlS2F!MkYs9L5kPF0Iz)gSbh(DBZpmNaIjpRWWiIzwP#7TazeO>Nvh=cunQ%&2Wv-!B>26arK z3Q+Kqc0OS1q7&`m;DVbrHh!RtJls=&VhJ#gDWLvBX2fE10wrlKChpOzK zYM?gAM$=Z0{TV&>V0bC9LY|(|{eCiGKbgX>`;j=0ZdROo5YZ5w!45l~N-ZO;t&5T? zIom}bDNQn4Y9Ny5FUF<@h;qzao_W5F3_lD*x%bE-z#ZOghGhh zwCHBEc8)SKM0-)>1qj6qBMuwGonbb$V|Q;EKq^ag3~T~E{arZ&ido{sbxfP%*-V55 zWL6`Njguv?k7dFa7E&1vmpSMd6m9 zxnY#*+S*zH(oei)r7BXNI}4)ag7ia;hP^X7m?CTc3q#@lcbK!K#KeZ?omHB5B*a<~ z{~I+g3-Uvw9GE;dZFHx6vQA3F>VC5c`$-L}`ta&9F)(pM$T(;JXf3 zdntWr)L$sxMs0H3rj1d(#uXnoagwgX7Q+=6N4qGIj)tX5wLD-K?h|buu(Q;ndq;D) z5e_$2cxFfHb_+sl1&1DEuHO@x>p1Nd)8drgM31yv@)jg5_^x1LT(e+!q?L#4rE zyXFJ4#?7;skw;6qf2&q&>_YhYMJSd%&oxL}Nkg?`8@&a`H#^|~+g7}SL zB4{xz9wXWysbh`b*JY~up~wdYYdi?yCzKE#o6cG&|2s9!3Nu!g`zM+P3$qCah@)^! zO+5vx(%X<;`zGQ5vsxzW9|P~`m8D5;R+j11yGul$Ra}zdf+ojZzEojr@iLOzu0noE zf{<@{PupAR)|4lxv%MAb4t%c(SN&;X*s1D2!9Sz)IpwkcuHv`T!;OO*^POOx1Ruv@Xxl^{PN7A|&$O}I34&Vx0?8S8)GDzb87{D&g2r2S zBo$V6#I_$%y=ML&wPv=+k$KI$bgk%M5W8ld0F@Lj$vIJz;j5aX5^jNE-b9`Pn!w7#sZSs;E{qIEe}&iT)Chgojyad0M2~T?jNU2!u&_}2sZNgcD*>2 zz%gLr;1%2vNcCQz7r?MM_NxqZBf_@{I2CuQpJoEp6VQ8p9Ikwbr1%KLHGBSnV$WHd zJ*(G|E(YQ2`Be~AVVs&cS_wrwuA^!6Tg=q1;BQr3gm?h@f- zi0Njxh@g?kJAOZ%qNTG*{wg15#Q|5>&M&OshJ+I0)MVr1yTXn~-T^!`vGr#~FY&U} z3Xs{11o{rOmjx7ODu(}a1|hWw>_>(x?5CKdtK%$(w1 zJSHd?(G9WF70>0<k1=kzMdQ(+3x5?#%W{dwxvqd5~q5CD26TQEA{>(;S zG>i}KvBrfu2^n`kUCVh#9+h>^N!mS$f++UX9f-Q|E7UX#0m0N!jk(&-10whU4@t&XUipO`C?-r9Hd`YWX85 z*mjmrpTGgO7K#wVOofwaSv|Bwu;q>UK3G>#Avx=6 z0zRb)n22?Czoe|YvJ7#?yTN>2OuJjE(CN1G&5aJOrK}j1mJdW=CGR-2Qv+VJ8#H}> zi*QO6`;`l#0fOT=+9X+WXK!cx_bj_^Pz%t}kFci|KDkZ8r(~`EZKhv8!bq*iNP~%s z{h51|duG!>>-PVycwIrq;`J7g$U=&L4#!}dwRo{u)K3GydV*p8&N$#}Go^`Zb>e1d z1S+@k1R6NR?-*o8ZzO73t&GRkx}Ra|p&wJ_{q+VUWmbp`0MoDc7LiCXk7-3wl6hgs z<0O`qTq2vulh)FS5ltc~CMt;|Ns7s?6QUdfS+p-ge52%^QxcU?AjG-sp5o&qP0iFK zX^n~dX;42&V#;4C*8U84Fo`MW{zu%5@Ln8^<|2PwQ;-TEQWhEVOAv73OA2tB0Vf<# zi;#{^qrzdqc}dq~^c%STmUd=O3%tu+J|0J9j&%h=vb3NPRWV^HS45=MTlbLd*~V2g z*m*^@V^WDYOPo__Ea8$;{CTI16yh4T(NK9xWYYmafJ-;rP+drx4Q`lsE_+@RKXW*( zM>#jJ`5bS93o>_je;OJbx8*9>#o;^LWOYVdKTyL3zcT5DGi!1ZM|Z_5QR+uIz^wx| z5aM{Cq>?&Zj2e-;2&CXpWb}!;lB#Y79i9dbx`ZT0*QkDmjzWuEKeTJ*iY-)|Rz(~2 zj*aqO1%=a2;eIu|*tmzmTj8i`c5-0YYKh9n{2`6`B*G zmvHVxOS&_gQI*2OPI#A-N0KfXj^ z^{Z4PZL0deNBR!dC{{7quc`&6zRbGg&geq1aO!^-12@!1Ull|31!aSv&9O|@p9Ne- z1CEW6mA!n2Ee$Nt!q3xuVQ0LEvA#8^Ma);n;jq8Xj>D#t$U+OFWbISpclYf{J>QWl zXonWz782IO7)Xb&8H>Y?XQLD>z_J=K4OTNrb(95n@;)(;jl=l|vg3^V2dh^Lvtg}b za7oNNQ$sYk_4%RAV!l5PEze{}%eM`)H!O69w~N6iF+Cp`Zq!&XjA|JJ#W++AvZLx= z3e1gXg;{t`(mMwC#53?WGB*G%WUj&BX>i?POjF}>Sr22zHs0o-+*)k{l8+TVzTUfX zHh?*=YUja+5-Egz$AEI@MbTTfw`+dmvO{V7COA{aF>u)alsk(Z`$~JEHIEo&!>i3> z+iD&;V(bAc=_JzssWR^E*7vQ62|Whd-xFc~ahA>xMEWK&oOjGr=Ae_!m)UUM8BZMx z@IfuCkt_qb&xnrVX$7F}iJJiB#mB!iOcehfKz%kbaS*3^OvS0jn{6Hbi6QD}{?(lf z|C&y{2PE^tnVxUWm*z#3V;($LrOO4ZNXstM(rFfR32osTKjmQydZz4ySqYB7- z(Gde_&Z!lxd~}K!aJa^*$;pGjji)ezsA_L*M|3LCM(T=)zQ{t3fY3Vl403e%v=Te^ zUcHI!O$z?da1>mz4jlmq6$zjcd#$!tTgmg)5|pyTH@61O4#yrm#0wCDkq7~Cf|u@$ z1QGaN3j7&x6C9b0Txo%FFyfX8fnvKTH6K9{c}zgu`;*UzS*kG)2XfVH2-jBZrUAUo@BTcAem ztaEj`=@%ss%h(J<*pp65w~*YMry@CQ8^Q7-$;7e~_sJr1R(5b3D&OD3JF_&aIzWq9 zFj5@r*dW7kJi3ThN-sYDK@3=p+!T~U44F=9i%2R-X~cCYj*EOu-0u^qRH~x4A{4H3 zXL86xp(EZ<2oU6TT;vX^8;VNPr*m|m5!z0IBba^%(*l zo(bV@owUmFJ~iU}9jao-DDrKA*yTpCTF8EpZb<|Z`UR2gmmt)!ThPwYu!t>FsUqcGkij&;X0z?RSI&W?`#)n_ygJf~ucpNvS6o>5<4t ziEQwp3Q1nb1YE>ZIsT;r6Xg2S@lpHK$S*elY4F~uNbLvAU)qgluV3yo~X zOM}sL5&>mFoUHxi5Sh~#_r%FWwRGZNnuQeq9$>vU@v?&VcAxUz`UJ3c(BfCzJ4VQ7 zi=ZQ$PEO06y?3V20~U;_hDihe;q?mK#KfYC_xP%af~LFw}5cW$bS6nY07{I9uzV1_aNipmHhGp^yP;!4gu zN(4x&Okedoaeqx6z8d(SfKi6UdV1ScIAXJRYa?iQoAO@Z>Q51i&7RzKN=7*?yyrER z|8W^X8vw7JmXQ1+Q$K9tJNDU);RhfFr$=kUoWjT+pZ*f2m8uL`#3|uAo#{e}8bp`H z`7IpHVk4bWX=bVCfV(X*ikChm<=C>=fcqs*kmTC=0?-4O@ zTbJh%2T*bW&N`|%2MDVg|J*twSZUVqYe6=d)$vm$>afZ>*q_5mWq6z=-*@19BB)A! zEc?1Dv+mimCtq-KaAmQc>11V7?_LtM!Ios}@EJr0aCTkZH7BC3C+bb?KhYH-C#iTFVtTN~uHf7Q z0)Kq2op!ura8+Mpl71jPyVWOYMyIL=@y{?L6Vu7Jo*YzS;Q)8AHWa@M^#p-UH30Gd zs>mQ}=wf%RW@@|W^myKtLyiIJmuhQ{koSCuc)IQ$Mdw=q>LQ`cg>y#MER?Ch6kN}# z3%6kLqHdRqG66*ok@U5e+ia>ZEGo*tSX6;-O-SfT9OOr`0@UV8E8t66RCJtU@~q+J z<}hLM0N;zqxo&!G%035@h~N4sLc@5`9RMOA4Q&|SoYb|3)dVtW2qsZVpb`6Vb338z z=De_cdie;BW{El@B0~T>0=Upc>{u#FsL$Ze78LeF%$pP6$u3xY;=mrLTyj_tlRkFh z3=PP<_QHu%^fv|5skA}6)bg=0fj)Ig?re3?oF>&6&b8|~$(HQAzTd@?cs3&I;`De- z8&}Q7HRDT*DA(a*F_Z|akh}U;i85Vc-e|$oi+PDp{mlr`6DgP{l~7XC!L~~A+AmdW zCa%ZVVy0p^Q(9Kfa#RS8EmbUJGZ-ulE@!TVd^4PRO@@!b*BB~;cAGV0leQHRtC%|x zMwFg?MvC}@O8wcI7iGSA%Trh*bQ~Q8g7FM0Tx(yooYN0vXq3-j;`wcHa`WHfHA58#rT zL+ChJD`IREms#_QZA34fC#_Hqj)cR;q8!bq=pY@!VNKKw^XbY11PaQ*TB8=YIKBpe z#PtWb>`fNn6KPx0AfkjVQ60759EP3Y99dAQK*wdu6h#WLT@KzMt4~=w##9KdJrTH? zkRpR6q3YNE(-!^dlDQv@FryXUyEk@v)Uq1zR<9Kf0{?hq^ub))+txx|@0Pz={m1A- zcKcq$fR8A~S75S=#2vslhC?)Ry&DEX%#@85$eN7bujRntnIIQf$`9Qyq5L>^9^Ta9 z7naXBGbi%Si4!awPMkO~d(fFVjGU7a-CRtb)3S_%pXnAMuaV|cZzEd~%dv%f7|=u} z4&fcuH#UL1V%M*688qJY5H)jTG7gC-RO-Yxq#sIN8A5vGotiuv2!$H5B2SQjk?qV& z23b6v@9U{>kz@8ZC^*j|j0wks?!}}NvMdhSH&m5ug~LU4&TB^SuuL-ojwIy(!Kg*l zIZA_lozAK0Y$WgN6wQWnNu_)tBJ{IWDZddJeXt-iD&@Ngn};Zx$Nh>Ta3|##Ka?U*1@7im zK*owQim}2$Ks}Zcvn1=JP*r0o}!n)DEY7wmjr}B3DuYGpvb$=sp3_Tj-y$uJt%! z!=|$Va!^;5zEps74Kw{ZZh+%lD0+wkzEZ>+sK=|0d+C+M#jDOE z{KsDP$CCPE-&N$5c^+@vR93=SBB=u50liJJ$d+hW%dS% zSt-g)g3rUWVc}EixTtUADlG%cz3idj4q^>OT_81KYmGzo3Hs(#^?>}e((!|>335s4 zUZod*z7Oi^BN#bcbQhr=iaLD)bPk$ErW_ z(|XYUKjdcw+B}bO@G+JbYTio6nPFs|3)nqGn$9dUwT1dHu<0m2wjHr&xylDr0L`%G z(-+#^`lZW+qkoz0k}Cc=HiID9>NkFdwsAp9IKFx z;oD(g<6KpT?$)yYH>`Oq#|l)Vd@G-VU*ow~enyESDTWqU2tTFt#p{AusGfl$>vh4b zId>S}zAl*KG`r9xLdm`^m^C8}{39f?lA03#e-q4M%lthfEpw5`xmmd){h^A-hWPAr zt-H36Phz=4w<~)bMn5tgXDWB92fcgF&`ASS`V6{lk%fmVR8XX|uRb3qc zt4yDJ?gIpvC^=K+F5_?KKXlZ0g$an_pwcurv}c+(Kc@W)4V(&}X6UK#&K-7R6Qqwk z2Li7U9#;IUvZHq^g}dzYFx$nVKD0*>$2`ykVV~+-BZCd~1*_jot!JP6pdpO{h6Bo5 z88ZuZ`5*@9>bGDhJ+IpYF%+=sXBgB^WCp5<7Z1qhurXYhpB?6Tn(afoXIJ+eo`5j- z9%MnHyh{sa=4uzW2U9KCK6Y`Qvbp7+xIPx;+&DI+$WpR;-<=9(8! zs(3WlJQ?DexIQ-5#J&gD{BeYe;b|~TGJOfiZ=%Ue0QksN&8Itwl z--B;vB6OUPZx$oNB$|#78NN|qjN%*BE}n00V`=l~qT(F=JIp&rGXr)s?^rlKEI*r2 zPgy+^^Nx@XVJ_&`Q&>$Q-ihmF^G@u0@QxdyI&X^fv6!as75L?Q zBD9;3Uw$AmOd@{i8Gcb_=g)`AlXYkLDT+)9)USYpkA0x#kHW z8;VG>z|I$+JEgx-91Ghp)6MCF@n-U)O{2QrHf#usqlUKwdwnTF+FxS1aycms1Sh^{4c;4!nnCFFsHWts5DzSNf*bs(0dv6=s4vh7V2$4fPU;Qnjg>V!5n0_YG zGZE8&)G)pJ@?fS{E#kQ-G1DK+63`(`KO9^zInSXIBKmij=}%?`>}aO9CgW}Lvk6U0 zt7lTCKX{l$eBR+4XQvwph^Xhah|62R{}jn49OBgv{BXf6iE52dovK#HAvC2(Johof zyQ|}o%j#*!HFSPtDx@82q7bD>Cow&H_p}i%ofqL!k^4yDQge`+->YycZ{>>j;iN@7 z5}GVbd;dwQs2QK2C0fjjWcaBSf;sbhd9#9N4-rxBp<3lemjn6 z$>EtRDXe?{;O`o?8`=u&_Ukm}vnjaL< zwx~@@Lx_s;cQlTi5Epz|IclL5;;rM-Fq~cwJw);vYF*K7YZTnp2ReFp_YY=GdE8r| z{yUKg3p1@f(Nfabk2J}dZT}!LR3f(hgkf6+iey`byM@CJB##S%keRO9t5uu=^uNjy z^gHQbP-uoUoqGN6Vh|b4eiqe!OS9i_ddtYT)$_}`=kWcMeUwG1PRm-XxPFy3-;lvy zqq4=CMftJe$~T01Tay`i0Xn?nF0qp0C-e5l6JrnQQuK~HG1YOAit!69KIZ+J!JJvI&+Uq zli5QiUdJxk6%iY;)3rt0E%-~5hHaIb?hYG2!=ab9$`6>y`AdyACTC|neCCq_TY&s( zBt`xd9QWodk&0X7QCLT_ibb;jSCecwu_dcmJq@c2{h$ggSw^?<7ERCI{Z!R$o~)UY zI#Vm_{@URsZYApz$b$k_t4NT^irttXll94aM}18+P{D(NA3CWHx+Xb-dnf8;p$s6M z@pW4*cSGkI_P@}Jqxu5z&{Xv~{4>JT{B$Qke_l=?by3u{hF<&;>+PIf9YCdZxFlYp8~rm z+>{VAtUoBEV-bcJM}%~GZ>Xcp_N7L#&1=6?6s>Dmg3Zg*z zRP`GrrXqXz`Mv>rIb7vayW|r>h;;8l5~C3`)vMGS`KyoX)ZUlncu~N5x6(M_5qdj# z?9N8hL-}^9z{HTP(-2ctH3tSVYawqwOVu`2IY?A#Cf6K6iglO&#l2E^y+*Y-JJ|f7 z<;-iBEGw!&sSfsuDk^0U=MHfIpAUmNEl4MnPr;NRoeRfATkaZqmr6LmNKtA?gARO! z29YAX*3Ih=2&7x)ET<+@K9qm*KDE2_RPAtOBF`Vc%3x9 zyjVU1fTBEUb2BZL>POM2`uC`%Ss*`&pCUV4RTK=td?yPbi8f?M5VS{?2-$~7+}RC?Plq0|vZXm4KJ?A_Zaicv1z;CJsyV-l?yt7!?v$$v9($fA#$JtYO zV$Xqn`LY0eyCk^@mpRl*4ji#aBs!!)&S;?ODHbqmfeoKS^ag;8(cX9ArsVUZWC9o z;TZXLDzEOMAl&AMFl_a=(Fo&JwhF$jyp5{rKL8o$;QuXI*SkZKs}yt_oz3UB5QF?G z&FJiN)*Sz`p5ri&^loQCC&mcNKpN;@QMak8;t(V7!aA$^y=WKWv=jpL`?%z-I05gf zf-BSO(;_Kox^ZX`K8_NT;vwqNJ6xQ&?X2Q5y9Q3qvnV{(FAA~sH?t?dI6Ga)ErS^$ zjd4&4LbidU+$dHIYY1Yr?3J*6K^I-fdN0s_QlhgYrQN=J;=r`LnE3Y~=3mRHP{J+j z!)#esvKoJd1;zF^tMz`WGJ9&6$e^i{ZrG=PT(5KWVYaEZ3CfvI^wg7)Al^z-pr8~# z!hlX?3Ftdh%(aDGs$oWwWSd|rMHKZjOqN2HWQp@Q)dVZf0+;sPavjrV|6~h6BiEa7 zfEcQV#fmNGz|ySfEq+d^=$HB}T&48cBAYqCn0NN(ou!NY`hKFmnqaK<-ve$KwuT3) z{TDSGO`!Ha_kU<{@NPK{)c!9)B!%sy>KF8*>X+~*dR0gJ%ru{crmxWqYrg8A;)5Zb zWhp#Go9m(Q&bg>|$W8;i0c439!|AY{O3Q;EAq#(??z~T#vFn%i99o>lHvIaf z#Y20iIe>w`_a0iBE;wXfK@sAP+9K78l~+bli~~hubBGS$-csRBkpCe35cL631)t)H z)IMlX-EL?{>)zWic+(9_E8}u-D9<)_W<9z#oTA3KyOs-z5yW{1(9(afye^=44AKbJ zFTK)0F&O|2y$V~VMknCgxV;N*(oBo)okG16qz2E@eeMXjr{(N^lb zM}ki`aWw$ll`D|a5ssa`$0sGVrA#It12V6t*RN3HVg0|ACsZjC`710rJ{}n$%-0d2 zsqBC+{)Jxn3IMhPZqA3YjPWZ8h+s7fMEVs5@kg@+aVvA6`9e=VNCGIVH9hl*!~{D@Xz!u(~H zggJN36odx4ckR9rL;lewuO+%un!N8EZ1O^Z&LG?+E#G~4=Rn?h{NilC>6>Z{O=C8J zr|y~11b!?-To$9z%fYiG2tw?OELK>@Dm$ z@Wj4-i~IN8K=PII(d8bQZb<2*7*uThPh3QJ)rT9NZhD0!k z>YnGwCN2lqjzbK|X;Csb!bP51YLD4WKI~e2qYtzOoiPQ zjI@o1`%`fp1nmj%_ggKm-3%HW=AWAbUUPH0d5*7!5Vk$&FgHgPLh!zhjE$NnuG^82 z5fWwGfCM5UGqQwBDp1c|#L9AWbAn>l$fBb;t@NSE<8FEk2vchvwD_lm;*4Uh< z3#3;*BM0bb4YF`w)Dk4l2;U}1p|qX)8J1q>BSVB2 z&;T?1Pflpzqd;IY{O*1}wXlqP8-qlgfvCP>8Hh(RSO6AqnG82JHY)Z3oR4I%7;SvA z7%mMy2PPB@77h~+5d(MlUa?!S#?`vuzxAT%iC0tA50AAlc0;~nuVT=`*rDZ|g`V{= z2H@e#z+#*`Ey>Ap#`?q?t#=S1u`*)@psTQ%EX*W5AW@~skVgCltUsJ3Z5GCq@fLbx zcTGX($~H|1Hv6I?Da>F>*h)&OpLz(C{$@gKNkQS>Ok&Mz z=$O1u3}mCd7HdJL^jg+qycS-aWa!W#2Nd9i?X?<--fJI*F28ns%8kNi{;G%VKSUK9 z*u9@U$Q8lY&-)j#*^d!s=+Aza=f(k`XNwokAAsE_+A{dR^*T|~FKny~%@nx#{n}*t zJvmPH{t|5&VAJ3RIz<+mU)5+v$TV!rK+)#9I7M}69>mGve8~8I&~0fqrQR)7z|CNF z+S|r1O6oUyG}=zP2!~8=L%{LzTFsV+y?eLY3F_{uUqULy2D-I$juT>E)DEb$ zof&yQFu^Q(;B#n_G;d&RjT1C3-Q?3Bax7{l&D5A(9n7+OFXtKL(2l*Jd1%i&IDDIu{^OSCSFh$ zASnikfjkt{@9og*R@wa4ldy6+KNi7dcFG{eIhuu(uxOOccyZ(kt*;$gENOC?Cz@P6 zyV+E=F&5!QtT^xbNJE3mb^2fo=|`KO7I&pILEjx?g31I@G8^HML;~k|r%d^aB8PMZ zce{2T+ZPi|&wN}xF>Ek?J6LNXgYoZbo}9p7-1ReB8+=HPlfA+CnhvmA6!?^;KvMW4&5Z>@ubS%TX}YGWzo7X)L8>L^rS1KSX8M$} z=9$E@rT|a&0R89j!k!qnHT^)EE-X!c14Nlfn*5rk<^{me$A=GQe# z&zjo*sHb)eF{$|Bx`@fYWUtaIU1rx-WX;hAU2&rWDToxcx<|)ygkYmRHhKZa6Fi3P zJUWew%cWRSl6dTst+s5WO~Mim20^~6<3sc(xzuxW*gmV%F%)Z`4(!_5aAxSe@M|es z9*(bD>U7zo##lkI>f!wMTE$0jz;7V$orE@ZQQM9*3RO!{*@tkac_RWhSF7`ZJA2>q znxKMZ*}p~rg3h>i=P=iMyP7xU8JGuJ%J$4NjQ2ofyzs&p;EBJuPfJS$4}0SGho(8i zEEX|`wTi>gzS*Igqck5`BQkn)VeeSDNat`6E~hf)ZA1U8M~v8+Dkmf?t+=xkb?27gRsTAb93HGyeSJg*nab8?(05a~}gnF;UR39W$}tFga4 z&vM#6K`-8MeuaK#jr5{>vTWm$Gff?9ss_OH1-KKq0vgHctcxj-EHp8`1l`#}Y6`BQ!$HE^t!Ys#7J8<6i`V*hN5CAe*Q74!7%o#NuwHvZhz%?RA}>ZFahv2I zg@FDY2L2yr3H)0JajS*L;5IQR#f6;}xr~vFR?F#^gQiOsgT>%tBlHax&&7w8d%Xi=ls5q&z*P9FVTa=J^WuBM81ee+-ARQc+oLQw98tN zUdJ!$dE&+QMGz{g<7`uM;Hk4;)r>!ZqW$2HX&v#)dZxnbeq2SXX53_B>kop;tgDwUPJFu#T5R#E58Q31A~wdyt~-&B22=lDaX zyHtM=HF>JgV|;~m9tW=Ac6wObA)oDyVNzJaCHe--@Q?I7$88{PWhyEeDSv4W>#5gM z)jx|26P|3FBk7c`qFFj&aHMLRFyNwY8+*dz`m+imG>6Em-HQa?3PylifN zOW3y?j*iq>gV0$bu|Fo0S*T9beIyxnXf9Tiu%3j8@j#Zu_=sff*T|aDnD$;{fW=JU zXbov`pput}olc-3KYh(L*}MrC0>D&^Q#*K$ytb>pM3dt_CXFk4gp-!N)KlNu?7$1=1M1jasXZPwaAzBKZNfA@O(N(!S5S2Kfz&evO zFgb{#YKJa~+TuFAB4<-+RD?Gvj|Wk5mb4t_Y~v&b)UW8LxZ^s1RZnd7JkfNj`d&Rn zNjI84{lKkiKjiCaO)usTeyy%)0lJ=gUek5VdvrP3R?$E^ys8;yf;IiQW14qXWPQwlthZ*!fNZoB9%mXQpEn<0AOKm)`b*D3M; ziwGH;3@Af^Oh;r5#vHW5si(_SV~uo*D6a_nvDijflg)3cY>Chm_GfDCL3I}k&U@P4 zO1GwgMLJ;x1%^0gLITNEra8s4=tF#t>SeDLw5=T)x$h$D3L)O4dSL^}l{JgXxR3s!GUINO)p&4B?A`3ncL1DF@MfXOv1UMgyfHT;G7+ojxi#5HB6_)HL?H!+6t;w6DcMP4wr#ZGGhQf(%A)9V{Cm{6Y(J#$q&gdBexd! zIO~XoDftyEzH&0K#!NBe%rx>&W7bh;E=+{%21B8zn6>ri!IM!%nMhmzubP=B(AKYh zr(WhiC&!5bk)^ioy#s+HzkwtGDCCz}w}sV@TEZjSY2sCXpWvRV{;noNQY-gQKC5w0 z$IY)?Gj2wXhhk>@*&f2nHxvdv8O+IlhM9;mQ6f(MFM0|m;N+WM)wB7Z7Ol?ThuP#U_(Rgqujk?Yc?L`C$HIB3(W^i za4LPa?p(lZoTu%Vi)tZ?c3iYqXciR{(lhj$hEY!@uhrd{$>=gkv{t7z3rw(9ckR*K z@Sq$g3PhG`_1weo5tsRp79S)NtFaYyS1T0j!D8brdt|bq1P0%p2S1B^t+t1K8LU=R z0jAtLEOPAi*q)klE5Rge@1x)=JStS!76nA|6lrS1tE7}Y_Fch$Yr?(*XxTg6^>}$m2O*14bzx51mSg<+Xzjr#}~(@iCXjFB#rx+ z7vq>xnRP>2gyqtEBU3a~!>Ia$dtcIGMNJ?3tg7m;{zOkbi9tMsQuRMr35U^QY_%+5 zTrxqJ__{67LhHh`4LNj2~3K@6R zxzMIJmtI+4zv?_Vnejq{>xu5c5VuSN$$pxl`56aTYM2zHH^>G9O}JgYQ2b;rnnnm1^IB=4v6SpPEs) zI1V?Ms{XUcNW;~%Ou=dWspb#`39IS-6V1GZ?}P)yQ6!YUbDLx?MH2m6(}Wk#uZ~3w znZPrM7umlv?{C61SSF;59z=TW9!fxUrVh9nJsH*r)1uoELH?5%r z>)e7eahDjV!fR~@)|!wVYLcucda8QXLk#)Vu_kTAqFadWUW^b~78%Cw8b^-MLffI` zp(clUqRGLF?e6==2tg?aVbEf`Gb(*8Ec@S$MnlVPjfQy=M&scu(LlKnI*+k#S6esF zMuIKU(O5E7qYj`9fv(3{Ee65Tn&_(hKsSm|9u(%9qLlx z#sF?i|9u&Mk7&G?8`FPZzO;9EW4?g^4a5JA7#~LBUUG>x=-vYU>t*@Z=#GoJBmc!6 z7q2F9UtXtZpRXw|j!;vPoTA~C^8!{*N{8r!_iyWuqg&Rsi`Agc-#UnAHYmP_AS7Ij z4)!IGtQLs>kkzYg%%y4D$`(RIh8U$2KkZ!cT{gG}*zGg3b91sNv8opaJDZa2xNMiU z>2vg9V`%fW9b_KURfVvP+kAr|ih0RU67KszK8`W>`Rh4tg|qQZqv4>s{J#Dq1#QWy zrw`&}v_~h>yRfZA=rw1#6;f&FEA_EOK)m3`=vSv302%GA6_w$Q|VcgpoRN@RY^%Z5H zh0(pkjw>lO6@C~-`Wv!D`mKZ3fQ93(EmeJ~wWTL`?DQS-wsKSgUOrQ0LDTLp8CfEr5 zM`KOEh&6RjoTxJ&?8U#2K_xMPcW60kea3niGj^C|b6{C1ebs8TA8BIs?57y@7ssO8 z2*%CpV;!t%hKqd*(ee7l&>}IN9~iD{ESN^MjDaFNBX7ctAeN}rmYSBn=VosDUg0CZ zpXCJQEX`>?Yd?)aYqW3hF(b$<-(dWMCmfM{$T`6^ekrU-JEQ@!gPREHxvO3~7n5?3 z6WMBsuJARBWHz^f-H?5cF3qh!il;J@Aokv0)oBA2-Cjng^|`sFq(`G07pZuRaW8y;%!7HNOZ) zhPF*|aqCkqS<|@C+@Nlhvem40#C<0163T9=;@)i}S+%{Kw;>BE;*XhqYH{GlY)-Ve zRPRt#M)9LiYPT*4_hE+Hh-G0~&Z>gpfD}(l-pu?Wt_G(6+asg_7CLq);J@SmSAk%} zV%^Bs-PUEVb!hkcZtbiXHGn$0YbC=4BP9k{h$p`1>o#Q9o^_~}x9dypG_^S&|2BT1I zuF}n<a$xqw5Ra+MQ48gDZN)ZX?KkaOWzA45b15h)-&d1vvNi5Tvt3% zE#SFdAl#;^gqNrqVMS@~yaoAaaJA@!lFTTRl4Dzvs%c_u@7~ohFnKg7H&wSHsF~rC?%VS@%y)oYj z<^h63VI6YeRJ0MW7O-b4KtyZ=`=5=nc#CIVjmN`P zP7d;2>$n*f1rXN|f8r|@=!|55tLjlXh`>fvO5|0o;4W1kb>9{2e1MAK(lBsp8)Zb* zo4A1x{g6kC3+=I6jBx2?CaXUpk*fcz+0A8)yJD%ZfT zMA{tM&+GkJWd8BzMO`}qtNk^UL{PO8u;|obQNoB8jTm)q?hG=1Jvyi*fCpAc0wlnZ z(LL~61?Mo>Ljg^v98kuHYW8vbuI#M~$S{MHdJIJ2uIt=J07C%Cj0hZVrE(3}(Y9#A z1GafY_L?xPWDVLwKyg;&9wTPwfTfU?j00nW)EJd0g7b2qa)Y>vXH;&EBYP4BFz}Rv zY%*1mXIP0YK5?h=nnS~zpqevkWmx{w=(bpbm*o;vF{L=at*_0J7DU?wLn%*6{nUb@ zr=H+b>yb|l4zl3{*8P=Ery0MnH%UV7g@pXPk&p^R4!}J)a^+i8O3-+0GFfq&XVS_T z3)J*urNLxM3z~jmX8}31SE-K6#h+aRX)~XUYlXJjlv*duc@kPG_bfPsW!`sEsDU9C^tz4DRw4N9~w;p21 zf1M?w7q6*KwD4SL{DB*7&RCOh<6()Jr}UyjiI?!= zQu;#OT_>#*YsQtmGFAPnENA>Xg%$eRG8zrquSxJ?_`-Pf%6G3=UV1)k^F1#o5H_Ew zJa<55(Do&*NrT^^SW(--f5(0-qm2YiGfE6?Mhhwn9V4hFi0YqfL!!xuy-;2prZ0;q{RtUvl2Sy)>*dcJ%acRS zE4z<#b<0x_OuG>;+;zMal!IM)XV+oBR^&gHTkdLrjHnYQ=+xdIyk7GChUJft< zVBlvq1V^RxJk0X+Y+<=TqwIa+vBz<-;p0!>AoAWPmgthhCm!E(Q3HOJsc=cs?vT?g z7ch{BZ_N;aW^mTag=y{>aDU=C| z2KEJ+7kFEqn-fu7oDo8JnSI(Kg35JKV}V1_F7E0iO9!J%!D5DGGeej_RA$RXycx?L zCW6}d=OfZ&93l9PS`MBH^Kw*`p>A^@zBe?$0O4@53;<&rhIWIwoFLjXV9W@cOQ6+m z;s#I{C^ByB)LevV5o2yUb0mR);#}Cenq%{74$IltS968jQDIzBEEbf4se~wOx7LAi z2HeF-cX?H+-zamfemdE{>R$ossCr1$g|+9Ri@u7KW4cs4hr+dd?mQG%N7&sQh+nCB z>ptCRN?7BIlzu6r()xZ>yyU4?0N}G$ZL@~ACa&(n44?&u11i>0^aqBsi)Gb}Y6&HG zL9dmx27nhCDZJ+Fg=-lAuiko_O5atGumNyS9IIQH#lIhk&&|!UiVV=7THf^d#Z+}8 zOB|LaSd6LlX+fpNU$k7P*=kqn(k%2-4gr&kQ_aWMzQXDSqlHoFswH4CuBU1FdL z8`*Z$Pej12Cx%0EUO^+CYI0^VKNvjj%3f>YMrOQl@x+n$l`|~lzcf= z{TE}==mb81Si>NJbaUQ5L`1(vLixt9x>E2)9~;P{Z}5dLW{JaAo@Z;WrxXMK@P)6;LbO`BKIVdf__}>D1y!QEH&Kxp7GUG3 zlpSvi#~V@^39FPSaXlv}&VmJ_9-VES4sa%x)&~S&cir`+5vx7t=8O-LQ zG_ueez2UiqFDK`6ZjL2CMT<{lHo7GkGaMAF=a`VEPDKU_ zPrS`NmTzYC-9|UcRRMd!n|Qdc4wV zpjwp4ZCVXnX}%`+1S{?Cmo?c})WA`db3c*gCf zNn8Ck!fLAeAx$K~;6r(m`DWMmYs#eJn{)R-)gVf7sS*@%xh|5w`-07BQQz0fyh!moa!+fWWHg38#JZ@<5b z185?gs|tGP!dLWV>v{{5o@?@FXA@K3GDgPe1X=t~HABW38+Pa>afIjRB0~-p?^T)U zc=(SsZ!3^EoZ8SQ4zCLxA2I28}qI3sp*_qKFZLv zABCmiZ;dtK^6d1W3I7QW^?WU@R!xlF*ByU#;9i;{B z&O5u$@ZEg$*IA_9l7>uW1*ndDVkhn9Z4hp0#Apt3fZy z#j7uw;c-?1TE2QYBX6?Q@{1ll6sb+k4ea`>1O*)b%{%pGjn4tMI~z5x05f@JciY46 zd8>hQz<60`l}-bydi(tD6~wg)D=tjTjolX^W6%(H2=4}@AV9Z{TBkF!;yiI}?a*Gj zDjH`f+qmiX5D>~gMv&pau{c+p<6gung`(S<2{)wcF*}DA_uCyL54}$C>qcvXpqCkjNS^6nuSYr+9Ds*4NXZF-+{(Dtc|u$dDnNPAab zW6Pe6c?m1#geHdcW5U%a^|zG7++m)Cr{l%QK*PNkY0Q4`?`mo)K-hb^O&S{(`1(-{ z(4$Q+>qDu;Ywx+DuyvGNd~CKL&U|_^GyC%Uw6H6Z*|Qk6d+=1r!bNSyWI1GbPTr3> ziE?M6CDPE7Il&V7*r)ZRUY6rTfyi`;+&>~BfQ2!vi6&X~I|$mT>PIy4NyDt4`VEb- z;V$dA!aHse1;4G6(Bfz`ANv*&B4N_I*rk!L8hp-x3Km+YmJ7~%aRQpGo-00*#<-+6 zWCGhhaP(i+QJ8R8sfasXIP#Q36ozvRt6e1gp(CO!2rmh1L7x$DPQaY1AbQo7kruIU zgvx|CF`(j1a62v`5-U<{3(m6KMhbm6pQ{IySZIz=Q%Fo4i-sC5<8VZ&;v;$_Vm&Pn zd2ucu(^qG)UAh=X-USK-(M}w0;HZ>3mW|szWqb_hAgognoKT2-N_mkf-An`g_3;Wm z6izg(;q(Ra4r^^r_j3VYYlF^Hl#ncle{m&hsYVylaCD#Vsfj%vjnh4C{ z+kT^vJC4Kmb)*-f3W2qJ#?tvAObr*H}OR4ta+Q zvs8ybG{6~qI`d!eqMoGO?Gv-cIf_0<74hC=g619K4S5!xHZ3S7i5+%%M=CI;q6yei z)S%ND7VsaAP=*hy=nyB>t#7HXciX5RML&vknBRPt4RZsviQU}_1K&=(5x=o4KN(q; zk(7inP67U4WTfG$N(S0~(5!I<39G7kwO+wMvYy3|rEDp*F+0XkATG?MU^8K^=Yd6R~u!woX zKuANkr3vgC3+IWt$H8kAtP$3ba$2{^pW4*#sV5lbfd`Vt>QJ!eAu3pdLR7=dF4B<5 zheB};x}qGP@S1R9$wh>B%OA+WUygpD;NXm!alKI<&FmnOOW6+QQB1AIha0gA;IB4v zToDaLeVrM2nQ)rb9jxq@&k`U3l3B$;$rV1?EOS6KE(nciQmDKVs^p-qTA-qBxbzit z5{WMai1@gSFY+}E?HpBKAOg%#B#GTusx2lzZ4U;g9V4J>9o~bm=rhYy%Yi`Z=p>wI zU8WEkx3fH8viWzji3qDznf?IZzA^VcX;lq4lgcBesdh79i>g zM*M*Uvm4W9n~)jynDR-?m_x;_%AfiP6V_4O#9^H$(?}%?zI5Zn(0=$ju=wYwUeGB- zqlHN}Aqg8DEiV&7wG#+()4FWAHb&mpr#l)nY^%4h={@P?Z7)<%3+=gMG!nkq43o;Z#8(EIruh5l@?L=+VHYSslY(v_Q zq3>Yxi0&Ih?dSE5vJml@#WuD_V(M?sx&=x!lb=ZJx+|qs&jN{dSCKiLj zbjGeS%x&yhiES!;1$H9}Yd|@gohok~t=eriinW$L4a+Z6F0qz!_G4qM{At8;2-_m# zoJ{@8$ViE{fj(>O1qBJ)3vr=d)xhk<^tR*-eK?CHrp?hT7MJ-K zWY4%O-GYAnzA=!7ndBCT5fvKbzl{VJucoRGz9VT+(2AiA;*GodB$kC=%*z-A5|jGP zF|yG5TI}nX?}Z7rm1U_Snfzslre04~pURSKu>}|mf5n}Zo3V9J+M)fcmb z@>UYT7KO#{ktDs9^pJs+e%O%m%c<&0mXO}5l#Nn))))KVjl^IaQTuX+ctrQJqa2ZV zscfgH3Tb4E4(8cN{3qHk{&bP ziNp+Y0THunX0usnUEA-Zp}?N8?MfrW#_t;g>1a>LQo3=TQ12`6MnG7Ek2$hRq2P%1 z?1~f~*vss1G`b}Uy*bSaE`2G%pO7ah;oRihD=yA6l9A7Yq1J8@$?{}~dkwF(3Vtp) zM}x-hj?4nX@kcbIG_VzM(+GS7WC=N(wC>|0?EXZ)h*&!E@4VI(_!WGw2%qtxs&eam zH?Tv69w?{E6;J>r6Uu!dOMERR&30T&_5T?WD{;#*rp+f4PQj}m_{8;}1jGHGNC@VQ z;FNyak6vb!jH!bnKzMu7zZt4A zWaV&#H(#X)w@tXQLRf%yt}oRD1gn?`Db)dhlvg#PswIhA2+kJ9k7}`1hq_-a!f&#E zp<(vxyLw^G+Izj*U?2Tyf_>y^b0A_-tf3>Dc}8~?hkf?qDrwh!Xts>|9DVnJ=nKg@o zWsq`el{RY8(N(|VyDN0(j2nD%A}=&@nJjL+#W_`5TSKaw;wqV11&;<7DZoLTD%Mqt zjr)0zQz+7F1J4!*_i*#I=ze?PSqzCY^5}XvaBWCFmf&eKLFiV}OFY62aX;MGI(5%4 zoTUQq%!AWWER1>%w*>t z_nWYfSEMj7K8^ynBwg@bx|PoC!9<^1l$7uTEc+QOJ5ujuArR>Tdi(5Y=yT9{Xx1^* z_`t?X++(8C?~7)I#5}sQkBR4I)qFKrs4}V5)pI-n6VTN8HV7@amQkDs7tE-72?+jRQ#U@)3q zsb-J`BGGGC#o50m4W)$RPt?3Y=6Xoxl3dkoCdYbl00teg94M!}(hAyel^JP|kT8iT zr%l#q$cCsA<_9C&Bl&=K1?yygU`vC%P;FJqbhk+NyBGr=t-3If z95gesy;KfhdkW5@7WXS7L6&n$;T9Nj2KLu-VOhkBIX!4~E5m+r2*AP25gI419e9dB zM-`hAP=mGv`haNlVKCB0rSFkUR+Nz?+el#@H^YP?io7G3$JDK`G{?dX}$qf(HYc24N(seu)La-P{N*#jmUk`0>pVSa; z+8h+Q-4y*#j=17K{IOm+P0)1o39gZ7evfhn~qs)1Gy~dv9;{ zWepb9$}_q{3Z1=@4Jvf$-!962KTVWn49m5^u%CLSofv7y(%|COywd4!V2Ir1dWUnk zs6`7sb(ij%`(1r$DCRWT=W%A?vF}d@Mj1%uu|bbfYr=!~;hDL+jMvj)tyWL`a2M^} zX)aoqcZ@o39E;iGgNqIp(uV0J4sDU=)@+&>+bbm@3o!l2Gt0!hX|i9*N>fFr?@vAR zjFEyOYQm0Sq&Op(Su`eykx4{k7j4-sDIWId#K8sohQP$yh23Z$i z=!5N#a)b!=LC!|$9?&C#5AX;AYBUz&cc##4*5C_>s1ULl;wF)kO9&~OrY%PSN`yW^ z7dI;5YEBMD!8ROQ(j?#Fwm8 zKmpopK~+Ld*a~SBT$KxB!0Qub1W6|tSwuY1X5rPD5f1c08${@g37=D$Hi^cODhfa& z1+LIQdN5iIDt<;+MsP>fS_X6xy+G7{dxUu)88*kybphRrs;o`f&sW$F=R!QfwIA&7 zCRW9weXq>c^u5JsuY^X6K6qt5jNdg{ftoF81!3($nTdI$O_LfwX$`td5VX z8kj6G`MyX_U`;lrp=S;*h1?$IDT+S2Ul{7>eur_sF-6+w-05R%7TEsx9%vI|{2dP0 zw1x`3@QmRKLNS`!Kk5rjQnBu_g*s!(xD(6REr9y?$OCL*gg!CcQDRMa*q%I4^tELF zLz+vr2euFkHvZ|EU}7{skl?7b#!J-PGar|0?3G8iVix+h;Su@~I{R{82#h{9I(Pe} z1q;SrEj(Z)M)P#v!iY7Y0Y3K3HAJ(Jbrkyd(BB&MB3hs?l*h1{{oOK^8Ls1d>zRIH zRNt({HER&-p{H*bH8&|KPvOTraFKM|!2#?Dy?F6xLXpw}D)A?AAaDEt8MCDMGi=H| zm?o6_yTxG(Am3Zh+!K?+y}c!EYt*P#p3xm*iB!PnQo~%lnre}iQByY_y4AWHzCF`M zn$Vf;htFhj4>Oj_DEjVx;nMjFfFqn()6VYMxe?Vu>V>U0z7>D5!urM;D!@b$D-U7@ zxHJ3QH8-p5-DY~`o_FD|uD#|Op^A=(*^R*OgrdlBhtyC~isGT>3JQizqu5c}Xng|( z&GAINwh1-SzSb1FqSR{eCFguV*2`0@n4Q(Gd}ycCGEvon)N(E`TfmJ~5w)H9mTLqF zax1IF%k7(cjQ&V;m?xY&b6Voa#ll9lowars~* zD{as@{(9!#Ov|FPJaI)SaUAcLX>M#}|PU>u}CVb2%oU60ww2<`C%~ zDBU8(6(Tk>qnvIi?s2n-`^N@!Uxp9*EKo7RFybbUie8H}8u{Q9Y#3Y-j-onwF`)If z%8e3QG%pni+4kJIqbB?o*?-7fncXPpz}led2{z-0LtDz(jGY?=?I^ldKV51rhq{PX zLgmr+`B6>3Y6^TcHPb(tT7V7mwFe14eyC4}s^-9Uh6unvVCiXFQk0ZxkYxqHFcjCFTPU44`E8Iv{Ugq05@M zd=NX;8(82|GnK#GiRROs=z$2)(wcr`GY?Qn_f=<#WTn2uSB;DTzV*yBUgNXY*sD@~ z3skBTxGkhgGa{FpsRilgRLPvjl^r#!l$S*w_y#hMnrEB^9>XkNH~!>lN) ziPGuD$J_E3l)$Xf@J7byPJu8`XyUGRi^8qo;whsOpqUBPO*%z7(F#Vlc_`U}SQMW0 z3y{zYp$+IG_O3Q2l`W-7ltiHRx&6q{xBV2uz7iU?w?2imw)aq4TMu>VRJQ-CQMA-t zSxpfU8I&&|2FHXA7Q+2=w~s^f!4Y$JC)W?lux8QUcb*wKXMd#K@Ou+40$JFk{Mu{dL>&2QgDD$< zJHsfO@IOg&hPM$Hwz#o{^sZMPyhgJ*YYX3Ead~SD=dq`MSAxDtoepY4nFnS^nP=$l z3w_~AO3l4i^SU5wO`U%EX6%;M(t5U<^3CRno-->>zuVVTV+b#ZJ1{Lk;@f(_OuX>- zvJUD$nKc4BQLQ|q#|voI8u|pZ&VA<~4@pUk<3Uus2xVOmG6O)2X{G1Caoq?Mg##Pc z&_8`ug#1l&Lx-6NVGUx)k5jv?tWy~``}O=i1_s#vmcB`T(8&!3-qC3k-M zl~*u|j`)jk&-tWO*il^?}xybE%p^&09?(l>?eawAJI zG$=5M5|fAXQy7h2Y;*moY<(2|W3nw(yDV1dQCmGi+QUi_Ph>=*05UL_9`Z|3rBd>W z-8KE&w5FO_Ft@)GoFeL4kFM}GK7l>I@k#uLM2(-uZU>8br)DwgbaiF^jZdL{!M-7+ zk9Y78H-TR|qi<4fmHmV#(Ra~49g3mmW8Ex%6-)_5XpNSr^4%1XHxrJYh4Ln;lDae* z_!UQ!Z6c*REPhS+tcsSI0y(cj?Sr2tA1FFvl$1*Lu2*^Msm@z>-g>Zmu)9rYn$+vX za)Xp+SxApI9oIV(+AH`5Cl-@kuOS0SvsACO)=R8c^NhBT zsgn}IBM6TkLggkefe_(%VL4 z6!gh2o;F8QpgRkmRJ!XM8oe^@eMrl+jm5Omb7ficv%p1I%(JN~feYE2-=nF`OxzOm{(Iv> z26+NPfKI5;klBnNNE{P#6T}jTR-C2?zDaOfdJ>ljtEgH8*{O{xJ*a@L;-~60WEO#t zB7_qP|L4pIlnGe^p+48Y77bMEnOH)rrJpr+QF0`TCtr*4LelGN7BKCIOFKdERBQtw@ z-!Q~w?cw42@!-S9_SDRkjTdcRxAy?AIwdmZL2@a(6aS;q@E)^J`K;S^*97&Ev& zp?gNHpvqnmVZL-gvfx0+Dm;;KUa?OzuNx8-u0TjY^m45ZTL_FZRUCE50!NKK&e>dO zoNL^mbV{RD->O4?;A}-CAFxb7#+d*PSnzX)#|hFf7b~l^O`O6P@dUmUkObN_wvFIe zcQ&CqRZ*~o;>WA2h5D*FYE_)LIBGSvi6hVvY^o%VETYsRBAPQ=6wV?$5=T6B4au;Jgd-kW|%e>T=(|g1J*>NT6sp_c=md8ztu^` zCm##ZTC}Wr5(B_0ezTD^6?D?=MQ_Wlooj)0AYz02u9P-D>pls+Y0qL+b7O z|3s@%1~o=+@BFTe`!dk6_D<)Hj^qp#?&ZrCb_S$YlxDDvT!~9%Buk^aicMTl5a+%? zHbHIy2gLFUG=7oigRUH`(NIE8RYAz$8iv$s{t7okr1&=wqDALBcZ4&LBEO*1;!Ng% zFnUb#liq=dd_pD{O4Lp@=n|^7?a=OWHz7wvWwsUeO^SL1x)4IaprOKIO@?nigA zD7yhaZ&{S44Aliz%<^V|ocZ**E$ybPBW0{$QutMSqDi4mkrG8&;VCgI?vtT56<`)! zY1QkPFTA~xBuknwVTi>|KGk)?AQfE93&->0s-8SlD7ulBmE`q}e8#G(0jnpEA|YeA z!TaUiHZgVvN!`&4U@|ME8lsY2o=?$0K%EFTN5Wr(y0+?oMgXKzYh{Gv5~*e=Sq+U7 z^+v?GLbvy@rv}aT7NyQYvf!pTXsD}33we{_E`a)@%C7)ApG>r+*AW#^>oBn@y-zNN zT!8LH&ZtGk6v+7e3b zz#YTVE59c^Q4(LdxD+HTZN?az74lWYPM0gfodQr&!d`em)mo#18XP6ML_QKmi{7Fr z<}w}6ASVOOdvIa%qY}iW#lKbI!zr8|7EyW zNH+w8f~)Xo76(YD&kDB=B0dNRaC8HSaxSzno@hWM968I0Q-|$!m&_2D!4afD?eeM~-J)u=> zmpq--PtE>W*nTsF^hA=|sFB>%g0SQk;rFfDDe20cYJyO&8y<4aDML;4e&RckeBUC7 z1dTmw5ly|;Y9CGEgt2CuN_WG+^M1q2>5i}e)6N2B_`ZgLX`dKqT+?;rN$ zw_q5)uVEm0Ha!u)a$Ujr-y)PLD{p32e!oL#mVFXEUsgtL|AL4Pbcp*QYt3HZ}_NB z4B1O;(N8W?-robUTPl-;tv4?SLO#msYv}6%pbTSYQ<1+-@qz~ zHAnftLp`pbsO|~0Ov_4*>4;_8Z|iHDOKW7WZGLZr!erP?R4Wfq@m3XUI$o>FwH#t# zyW1mob{iM1x+)E!*wMkE+RBPRus108D7xXP39mG0n)FS*Sb@FyBGPM@Q4kNcO_79J zC)2|FZuczuN;S7en6K%^cAeLI!!})^4FFcwV5qLLBY}zSU?| zU=k*aBGmvWH%qVpqV73+lC-MJW!QLEGvvJrtk>YqO6D>m%ffP_R%tbh6XL3_cefc1 zq^eG+&H~+8kj1dn7Yup!%a?nL@%47))*=-)X35ezaL~Qp*m#{qg057u>`*V z96lKa-4?4GzAx;^FDBqVo;L5(3pK#1TQCgQPv7&mtVDCYNhSg5Ss)DA(NHi9wE=7x zhU;h;7|x`-4K1M3?s*s}-yO*vG~@3?xAV*({sevM!;yzT=gdo@@CvM3<+XNzV)td- zd=P6B7cO>k17h$T*iWz387rL$rF5xm2WfY!akO#kC=AyJHulj!edS0v>BGbzITHR6 z&P|!Q;HsuUIE*w16x|al3N%nI1eL(RlkmXFc908@cmW4|R(&wh9-HLfNELFhUX!^G zoHPeWbr4MD4<|`<5GBfmb0C~miWN)0-pmKPEA_TtDZQ~->J6}~qom%R`3&yU>_8pz z?uAxlK7+I0qIu-KvY+rI(#&Vj`IpDIdK7sMz`cN&slK5kaa#w~(;H90DW%vN)EsBOuaF+BbL`8d>*5IzU<{dRjf=^mv~pm)>#wo;FW#2*JP@w_q8zr@moO z+XD854r@2Z#Ex$=Xf2kAfu6Gu#As6$U|ysMfkj+ zL$P~hDLfMAw8#Bw@gj1XVCt;KlUYAgLu=$7t=HhY)Fdx0Fs4oJf(outa#iwka`LE{ z?SH2a85Yg>-(zXkLR-_lqzWaU?=w?ftJ?wo@bTcm{LI?H;6WGX{4U!z zX(z{M?Bj@#f~z$|PaWvP3-)+0O*h97;@03|uWpt}(u@0_g8qZ5c(Uw|CN+vlDU#@# ziUSEKM9QLZbc2nM(I^CCx~Sa{d3%)}sF)`w34XO^fov zZ%bDWrCM-)uHqV^wkXqC1BKAoZez^Ms`?`&eu~?mRBFg7i!^ED<68W|*X=TLnIMu% znVRxJmsNL#lI+4eBC+Lq9aS=AMHkMH3$xT3CCVk^(g=%V!Q9LUo1w%kgSo?q3h!aX z)^2r>Y*Z*0JFUc7(S_@Wg)kByr1Zc7Rv3BoUW#WzXn^lSfNU@qAb7AB3Z!E=&1@mn z3&#YCL}@6T1AfLrS*4}3&?zY|i(vFsWGF7Lj7dR+%=op2o%v3C<-LQXOFdyrCkwrR z4s%b|y&ZKGFeJe!t|?=U_@OrqWyqkH2R4FGA1?XZd&n}Def%#q;j1~YqV=v`glK`$ zyNwSbr$9mitInamz>(&sUf}E}wcXt~79hU2J`kL;f;a1Zr&h*#VwT5xh%TS*3n*#O z>;=?bx!o;D^dryoeZ~c~=E^wf+TN61PI;X0GxE@9t%aPM1t~&*m24XbO;7JS~;(K;xeeaTA|F;c0VRIAIKlV+)>?I_w)FrB?#( zhmx(hHR+^ZcxL6w42eR(ekZh4(Yk;X%0Bz7-f(Ier7oaJJ4CR0&v)Ks6rE5T4?ywE z(VB{H?=SpAsXvd8*Js$s4dqWy5@|iQ5YEsDM$RJ1reFx9B&iDlB-cU8lVY7+|6y0> z>Mq#U7t2Ty104}38*-ko?nf7bTqbX=#hF1H1*Bl1M)e}fQ(Y*M>n#TonD{RlkIB*J z`QRz|b0FDCqfuUlvd&W_<37y4tT30Gh-;`)Bto^gF2c^T$V*aaAz=&gF%iQqn-uEH zWq5wo$>H}>t&WU6l!sL7^P(*WJQOM%3@0ZgYzf_p?T8Go8Oc#n7*wA@HdEGwF>tvG z6b5U>0_XUF2OD!uG6fU_S|uz{t483{I%1gD^~Ag^!WX|m1^+Qbldg`tS1zQIb@XpT zWBL&~`|D6=(F86fn;$j=t|r3-eusy4EC{@>eE>Ns?j8`I0*3$Afbao%$C#_}M*|>t zXfFWn^bxrQf&aY+1RqORs8Tv2+8j+`|5(#YrMteNvDo>EboR0ZQd&I@1J#p5vK9-R zp}QIej>iXQaTW-}^x~PldnoUftW<4y{oCF2&!)NQugw|gHa@Y3eIY!1wJV9EX=CjG zgxd9S0+@k2eSR2=$r!d-OcR3{bBI|G*-zuAi^ESX`X2h*UGVKRr|WZmF^ef2qvJ&@ z>kRRUR-I3rq*SfVa%Hu6SyJ^5(!7$?LK2A)baY}@dOT8|XqHA$_k`plaTh?zVbxb5 zQ+pkjb^e^sE#XoxWFr>Y=qoKnnke@@n3=(Ehks&$J`uvPyEEU}1&UfO&l`8~J`?IV z(GmUQM#fMRV;SQ;3KcyNXaN@6G#2RVX*Cp_;)e^`kEIFPK6}|MINay4la5Ir`-^E# znLTWXZW9xx1=j!{#{(ZNEq*CY;BO7>FHIVq@Egyx;~9*#0}O+4a{zoa18}uri>h7O z+=eV$)4_yH+3h6j#?$GYfJ|HTUVPRDdy=`jt+d{5%jH7-a(jnb85`KuorFd@YXg6R z_4S*}K6ULi&~@>@SKcT+^vW9v>}H?)h*skb{w=%NVUheBF-dHgQH+*$1KV^3U#AN* ztsvUDM#d6`Zc`c=XyC}VWd;kUO^PkqZQ-5gVyMM19E26o!|z_Syu5lqI5{y)?ZF%jMejQNQ$G3|Zn7d8G2 zc!u^Co&YV+BTqa9s=(R`<7yKYltKyK3^G;XW|3?Rl!KNW6vDYpL6oLSF2syNrQLwA zfd9(~dkNk7E5AK?L!Ui8e;mQC1K<6Jy5rk^-1=Xl3_qAP#o6)LBbDiq2@3yXhX5Fr z!qb-#+V}=C0P-1=mjUg%P$*R99G+p$E|*4@;YtMsi}_TioIfA9B6zfj{0a?#Z8H(t znm((8r_IMSiwZ5tTeZR#a<2YdJa}~ZJ{+pqrLdUvbkVeWV= z4YD^A?4;A52SL$N*KO@XRQl zIIlbG!KX`{uPNeu@zkKyDW&&CjtU5})4{_;rDG)YTaPrU(<=z~Mz+72Q_$b(z zlkomU56X>D)%k5z%}1hUdklscs!oRpU>QS<99JzWFQJK=#rth?)1)E_>_4eegW>Xr zF0gTLg{v*Ph(h-;(NL}eUawthmU1YXuu{s<4LRrb=h)Q3km6ABqP9snY#{&P?i(2ygT@wQH{8v&QDOj7YOa_&B~_4C zqt-4_`b!M~%IM@@xK#Rn5Dg2gcset>PO z#!*)?4#*o(Re66pO$hhdtFxGZ@E)Xg_w$sGeH@>m+Mz(ZA=N#(akS`BqvJ;A=o3Ief zi}ZwNTFc-n?kg~}RnjWq!P5eJGn%B}G@!UE)49gZ#vjx6JFxMu``Lx19EufY4U_R= z<^Ljql^oM+Ko0pol~5%JhyXUU_koKHyrBw0T$mazOwdPB0jGSC74n+Sg-GZkbaK`R zu2tNn)`Fi`O#wC$B+b3*>=Ys=j#bl#dxIX##SvKJ=atN`*8)xl^46kL=2E#?(o-?V z!IMJ*Ywe+-yTWSz$Ixo}xSOZV6T_k00%F*n`i8;cXvifaOYaS{@)j^DbvO)6UmHV* zdOvsKT6$&?&jH6{_WFZF{yzSOEv>A%GC4FzF=1;B)5gJvnMVcmPhSc9CVlgERfA7w zookb*qmBa4r7)AH(J3+lbEUD(P zIzwv4Ge4^3R^O4b@CaCcZwta4G5O&_ojH zJWu_GrX2+c>pVUe1=T<)E575=)8-y5yN{AAZ1!y8!Ef}a)`9ksx13Tfh_1rm!{F*> zy~QjTtcRYy9XgLxRhwxN=H^?=0Q45?yu|~1Fdi>(Si9Mr1cT(mncSoE;ULF6r5>)7 zm-Aa~PC`toZ>6bn<9IUh);Uc>d~Tex<<<3~+YA?KaNu4+@n=#SWLBIPG)q4r{g^`^ zmA8s?La6psTY70rK6IOkgJem#Sj+BD@!@i|h+3W)7CKO7dk)EV(G-@fm0}j&U^BU? z{E^3UGgDJznQ9G!2d;D}A8(^juNT@`0dQV8O=aoldNE6l$M}=6OnYv7e0+ACKh3nO z;@9-lRECbFS#;J7Vs@1#RGga~%PgP)H_RgjW(l6-)dqy9&FuF1$8(rPwRnlr{vyrq zN@a~Y3j>rDZ|LjnLavRx!V9#JbBX%oMoXf7rJ1KcvfHTRiJ|h+1uDzk;cB*0ST0uP zMvwAAe4^ZxSxn?NZB!#&sa0zAxlyQrQLCT+Mae$z9U zP3N-hiFS2tvR&m?^&%Ic#reFckAqK z9l0HMw@$*Xqfc2@iC&LUiF#%Z`8F+RC^68Lfg>}Aoc%K`33p^aVZAKX%pt!Ht?U$B zWRM|*)R$F?PDCbxT7b&OE2Mq|TP66!tcZ-z^&*0tplh`dEmWbzBFiY>y@vQ9WV=Ot zPXh&6f@PEtSka+A@(s#lQ!G%Xh{De6$R&%2$a&dP`|ThB394VqAOjgRwW_b2a4Xb8 zZd*z{E3SNaeqf=(K8q4Gn^Xy=fZQ1)r?Y2r!OOU|g6)r#z-qCD;|iJ>8nL36v71A5 zAlvT9yhu8N($Kh?3PxV$+bn!6UZX4S2I)>{%#kzVHR8IcdN#l-!Wi5r(#2E-h(N|{ zPLNAZITR{YC$kppqX@LBtX@K780A|5xG+5f?UbLTrUS)TDb;AfixRHS$;oG>OoBm> zC508z4pCnH+19jZz z@zZ@-Pxoch3!jsJmgaQdEz&aid9E3_6WuNzoHH7nGp1Jt=ayU_cIr5?eVxI1dz}%Q z*g9{=*?E>O-r!(y4P$Vy?s%V_H`0NKw#JEzHoCsR; z#@i`DOFd=YGc%ZPSW2}Se1a&0`U%bW4FJ{>-Y=b!S~>AYH8NuWC zh&2_1V3$J9WLhAK)7e)qPtWYATAD|9#wMp{@chW^bpFthLsLf%AIctnbPUBb*)DZ? zDnGrl3Fl8)d-HVhgFBNeug;h|u(Z3?C}P@M9iH}He?za5L5)Rwcb_92g47{M+Qb$F z4^gDj>$RTe#zmr?0~^hLNQk%fXMfH!Gk4JrWHy(!XfGoN#MGlRk3Mqfkw=cqW{-@u z%!ZC^qo>K}{P^(b{M=VPleW;g2XBEm?9~o;IvQDi(4L&cx_Yc1SUz+kD!&-p< z6FxU|!kwR)ntnu{a}=e;?mi@FBj?CV`DgRbAfjeH7d)MxZ>^Sd!Rh=<{QE@y>HKrZ za#_5X3tr5hDIhBm3z-Y~h1M(CQnR`B_~fLt9*EtNuhrKlVb+G-0LSSPLQjzWXp(wc z!W14|ItuS1+~Y3208_)#(WOJvOH;E`Q&X_)uC9$yIU6ZKv7e8%2Vsg>MRpU^>BgjI z56wO*c4^rt!h0pChM9NtBkaikCPZkZS?w2gHhttMqQ4SdYa;4giQa*#pADg^_xV>k z+Ib;uDtH4rVrd4BYhQgy)3#bXaZ{WzOU(O7y;Ub91cTW@D9IpppwB%-7JpV-KpJjl zZQ$S02NLXa(zF(^S))xtgK^s7Z7yHbk&mp58xPJfgal+4GOaDRD^P;qs@ONMLb2(O z{P;LWM!DI7(Te>ec#Fa4WsG3t)&L_JZXjw+7{xe1%@Lprj?it4a>RPNCQVl)U;r8v)6hj)9b!#?JG2{o#sp8B>Aq2J&3=H7Ju`I#-i>z<9HNTWwP8*8k3%r_ z-tm?J=MT;PRPgiuywxQf3#{;)*g5|vGC0UNhm8?`ndY3kQk^J>9P`j(#;oPV~mHtZ+7iFg)3Q>ghe6)mlIrR_h9m zlbNTEiwgzFgD;<6dM1E+g9xHCBu~DWSadpa(;ONS$scIfj<@^ zWrv@2_5c(mX7S%HPEUtEPQ=o#`(cemH8``hhm?XIjWTOrB>e)EtVUAeO9)IXg9elS zcES8`nqa=$2c}wXGjJ!mod+b})(?Y?HjfWMQ5r2n_2QX(JY%t_#_Tg{3xx0udTD!h=bkrLiH6fT;1?8wvzsxy6% zwaF0&{<+|>{M4h=m2mifBkD4HJ zlbbc<)}FE+f(TW6KTl9o76hXZ76vHv)&=M%(gf%R0w`|*BOiDGiQi2Y*1SaZm3bK^ zipWpsCBkQc4gAn(S`>(+lAk(W1{}eK3wBvLq>k*%0ofa@D|_7LJE9I1!iDRhax^E) z13U+Eib*zV6O_RNicDsFyi$hB1serQ=!(><V}a&->zR8rmW>Eqy3yV3Y9^YUI8_7Q*+smHm3ZFhjys;0nDU!_ zAx);&?S-`I{`ZiX-vA$bP#({)tjQaORvz_?$D8yMkujdd>Gw+y*gg&$Mh)70O>f%Ln2BoS89AQOSQKIyjkU90Iah;+Vx0sa zBp9P@17A)2fK#HBFevMosHxC^hdDM2=X}RQa>b+|hLCK63((f)7DZz>9#_kMg-0>y zyTSuu2|m;h+43#zGiH;TCD{0V6mwR$NnoQeSY|i=u7ly_M52yCwo>;th(DbyBEBFuL6l9KE|UHL zJQ5eZ)|VyyWi$gl=Q2THktg4&fXr}GL$27{JL+9axoWujMKPKdj zt5B!VfTI2|6eYu$drmOZE!`MK_fZ%KIPO^5;PjvHpL9WXqg3m4Wt zi)@BP|2Lrx;O?y&98YJ#4WCJ=Lk~=mTaE3pXANVw_Sr2VV{GYm z^`LXd^Y!%tN-xvy;u0aa4^qdE_nn4w<0#S0fsJ|m6JyvY-3F`1OJ-_M@F&*x|MxOz z9~(;LqoX>}G$}f*5knE4V?M}WXIbrOqf@zBl!<0`0P$MIE1?I+q(FWQ%p54rA)lFF z1M5)PxL0eJs+Af_RLhjFYh~Ktl*JVi`z5l=az)%axDbR7GASU30nL@m*U}%%a`-cv zx|m_3S#PZ%pN)-W!y{(;hu+#cgM5u;dp^OA_q2IBC9GQjq}FBMP+44H`>Iia-cX&Z zHwO>q;(5)7+zm}v!w@2Mdg+Dxpia2dLb=6oVqjN`I4F@-ztODejR6Ou6$(nG8#dnh zw5>VziFXv|w~yG?xYB_Mag%O}gdm}EU?SaEi%xDNYbOJRb|0M zB=ZG38Xvb)Tf!fT`7HdK7WVC8idI}>$}opr)P#tLYNc(3ShrjLd!J2Z>L_;~8kXZe zN8jyG-_h!%bXR)SZ~`?qRwsQqOjQHrR}s%=tlmj*5PW8koQf~*x#pS&f}>O|0|o@< z5{-x?iYTJ7O82$6W?fay;&L%IE{MFfPL@OTfah)+M9zP>gj7;edyz~F#*A`x32C01 zOO34xYTC1$qIxA`4CSMejj6`qlB{U6#A)zyL49V4%fIHh0^yP>? z2KNHLYrte8rIz&;cpo!f&2$@?7tTKa?D3=Lmd-x^{JEuLC(puZ>)_d9qg81(4rWfA zeEI`tk1w4(W<8xeQKpVgKz2flWAo?c<@;lhF8S%?!Az!%%l29`s5SCig=Q&VLZ@QeKfM>U9Yd@q&?rNHzKuRc@xE0mI#3CT6uk` zyehwt=?fKb8?|!PjGBW#7Pl3B5UNEN_$#^#Ie3oFL~1EIpaon)m+lle3nSjpi|%P6 zisxW{tGa#=lR+xWLUV2uzmP97Q?Bw^I~UZpkZaLQ?%-({_l_f4n#E>3KRA{ft&i%w zi@V!P?#tcnrFa5SlXsV^O(Zga(&koMQ@z>UUTS;k)0*wQpT-dwrEd4k&G;#OvMaU6 zJ=>#mGnTb8=Ii(mC;N-r!d8JsXEH&F(#Wy@R+SS96k#OB)C-%9^P_9<*37RJkf^jh`U=EjG_04` zM#af?8VBABB9%8%Y&WPkRQKal+f?OmTl^)ZNfPBC#zCeg5s$o4=d#mNxktvv+>F{H z?>VK&-X4>mQg&_nT%-U$Uw#ECz4^alNdA+@!UV;bcqF=2pNhF;)tt^xh29g6iWT#I zTE3@YWlr!)ZDnrCNQ%4YCEVhFgrfA&tG6^u3cmvmHjt$7FEr~LKvJmxcbWzMmh2}y ziGH&8#HoQri zzdCZYJ2(ftOJ`#aek22>c0N`aVJJHcYtiy{k&`AP@gj*Q9gh_H$(ze|#|c&2nqK_L#PMn~X}5klZ4 zA@V+wS|`$nk5+|<OQrvGYi0yl&ZmPP0S70gC~fi^2|(eWWP2?XRyRd{JD`O`0>5{H@I@ee6i_J=#hqk$nJtl= zlRcpBE!xOA*~Kjcz_1bXAGo#C)`*40-E|UJxkBq?*v>AhukeH{AiWcAGA^NRGd{Z-F@$InifoI9Tk4IQWcIODIWI<6FwwrO?7eNE<7NoItmz6AMG^8R~zTc^Hgo5M0gT z@IkuYN4Ows1G*acIG-ESGlv4IbpbUG_Gc=PLwC@_j*Mxug$e&YgM)~YOz;ejSmUQK zMo;X};INtp)7;|mKp)%&RE~+RRUSth6zRilEJ6`+P6ktC&?2`T@w4X;yW<<`3h))AGNwxwRwtS$b2L;m76yy5J9H)Cb-f z(BL?XQRq-iK_FfROR)67vW0X~Kv|_`slgW*?2-(7T|8FRC_!~Fl~F94gH1_1c)bnXa#%&0%`FgMM=rcsD+i2u>xQZC|nd3Fdlpp}>o zA>ot}bqZ7GI{ukGLP8zePpJO&5EPjKlaY$i!YfgFrT&!vXH}-m+b46B5Vm{l_wP+etdkqR*@}6$ma}`(D)&`Z5LN3 z_&}AXIcbO_oQSCabaF_}tdM^*8x{~tsZz@r?L0U}?sG7CP~d|IC@5;}yK^EPOd2IK zwH~24{G2As-X?lao1=u=A(zzxD3$K|hDL2f7%D$@wMooE#?_z8Fqpi{yH_mmdNlBi zIk;Z`dm*Y$q}P8orbM)0~0TT%{S(1SmQ*`oJiuFk2N$+k+E^7#|A!z;CxA9zvO`p)-1ZTpeBF zSVUpn>Sz-NTjG@PFj>jN$GwPhLv-uAhF~^0oWUE7Q}N<+Xp~R{QM|a3mkV-!RQ1Si zDp;ImZAtc(6>DZUL2@%Hx!#x?UEyKi_o5e8{E?@zss)sPCwF^zz~NA1cbW^xaE(b( zc4Xh!lPEfc;zxwEr&05Gi=j1#9?+x`sfjZo7tE`EKvZ&Us+PR9+|GhDX#5l8iAPwg z*DmQ>d~O58o4$3jf=!eT-)wEV5rggSVG3Jt9L!bott+7en4R32h)U|nEo@K8dcvpR=KgR2sgBM)XW z0OI1#d*8@G(=ZCcPZ_C*8c422?nc62%65JdDp|L~(#bV+oI$2HK3v#mi4GLF6xC(W zCJwQqlSdzZ7~z!4aKg8dmpHovdoPtJNV=Ax>sT>kxMu5H&&Fh5RH!3LR69|f zzk(}kNxoNf7)YB3rd?+wp96-c%`BiNdvrgfXF)a4=bqUPF6~c*;C9)F&5JCL@Xx=d z8HmCOOZ$7_V?cY!f>1y5%sLv!im{>9@l^JvHLH|590sQM4#6obNQUZWnE9X3+-N9n zWr31bkHbLq-XWN$1d@G0n|=J)H>_jN-RvvPrjf zxPFCVB`S#81$v)-7Ae@>hXw`ibIkE$X<}lZa}ZjiI?p|ri8oX@98%ItD{MVhxx~92yp=s!J9v}OC6rN49$suU>y)|z-zs6s41*w+wyhK z+s{AIWT-@9mqA48`d|71Evo-(*-v;9sp|R-dHbm&mk{;`TP2Px&Q2uHEp}>w?=MaW zNRJc@+(w)pluJ=R3)wqF>@sbClQ407e6`ppuTv2dn0mJ<)(Gks_uDK%yIwDm2eSz+ z7R`c@#bp$zN7`5Xb9rHB1T{^Y$c%=%+K72Z`3#k#XJl)UqM&i1iepaJFEYhZ&3VW- z09(j?0hO9y8`(lw6xj}($s$w0>hl z4D0P`ZLJE`S3vIPbk(A;Zyf~2#nI{!R$dvm}poTH9Z*jS3lw^;6#8v~3gGvg}a zt`ztVhXRxs7<#JrT_^Mc-QN-F-rY*}ezf%6ul}0eL~43`);^Ss04>xd{lbR`r&iOg zE>SB?gdb~EFbf*(3wSAL^a9?O2T5T^iSRzj%hTp=|Aw^i)~iJ;hR6JzNd=18L6lAX`cu<`d>`76eIJ$HE2|a1Gm*>6|Ha zJx$n_t}t6RbMIX7p<=NbrC_wv#OizEjv~FQI%`}piB&c30EaR71l&3~g$;!nzc!jni- zob9~u5_x*4B$u295BDIDW=VLXD;M(k;4HcD^kC;V1n1R@>lCLB!BQGWD3Y(vw@*e; zyMcSQScrUpE%B2YL%!TEw}@1TJQsDQInVSub)3iO_~%0%ClY?HyXT$Ce^O11g`f8g zg$6Up;rj8QE1rW`APvVsj;>VJuo(Y?uUb*EV4hjBaR~?-i%D2H>94v{2if1?ni+W28 z#c&obvS%Xc_>?C#ka<#t@0#O+)cr#m#}1zE`tb4K!TikH!QerWUe!synpND3(t9@B zkbB7Y4<3DBcdO#+fhyc_dRB^7!;T`y21*DgSsWdMasQKZ8~q@8dRYC@yhbrEl>pW^ z8-_=vgiVBrsM)BQczP9RqUjLsLs@9MXC`cs#OsN#EajN(!)QcKj9bkmcu{IEM7Z76 zYuoTs#Nj|#CRahaMu9~PscV?;kaNDsGESBv$0L<_4?V@F7_A zd70pcLj5)1##q?SL+)LYnVRaOBV?yNgA}SSkHE=LzLjaP@b>u9A{a)gzImBEHFp+s zlj=7-A9v&@oRCYw1zT@UIxyN+TFj=hRVvI)FAUEfb1tciByH# zPs~I3M6oAa9pG{V;aTwF6TW`Z>kPUm$+;@i8XpGaW6$bmvtESIt2`d<9Qt&5a*A$T zgf)k9x8hTj_s0E#edB_BGOMUK1!)oaT^i%#CVSe+)MjRh#VK|QL2XX2v`>)^l?f2m zR`;Q%xQ_cAeV+*R?b47+WiXj1RtRJ)4bB_G z3PkH#rB$p?#hv2@GDk* zJvEoEi z;MZ{eh9yv2`sjHNb4efE1Bk(A|AiK0nzEnpBvKWFe~L_AxP}o)$#DgUG1qz^o2AQ@|$Zh-8=dEEHHsG=&S(WQ;L+?kNKJEN$1hI6Mp zQ7N=V(kZgnzyUVMA|8kHEYAfP^YAfD1%c>7aFkx8;e0*^7BxyQKa6ShePn%mqglz~g{Khjzyu@x1;cycBUqWJc zii?RIx+Hg~63|bddf}|NsDoVYv=0>{LntD4_)1%pA5)daWwO@jx8*4g)>O7 z3z-k;zm?%UQ(Y>RS5d1OvuLp&JqLtmid(gn68W(rIvipz6^w*cAN6G5^GA_?><lQ2mpc*gM9tGiLxKMf-_SEPn?z zJQARh`-2{A3=dW*JNsO1c>@Ojv5YA6rjXW0N3g$E8)S`!S(;!kBOD(I?!mx|aEfF@ zcveqClvko=8<-J5ciWlyiH2luT&-3y8++Naa!r`YFH!Lcq+kfI6aYl#!SY3paGDzh z=26Y9yb`i0hwYidx=qs{go)~!OJvJmDp$+RIlw`Qd$yMlC|aKrRA5R|V5UM!6m*me zBte-LW~ETm;`|1=kKiMn%fVd(yGfoeX|j@9t)6>b5y+|>7=Y?>fKGAyL+gY@5zWoQ zR@NN~xT>P`-Q2l)3#obw+jE6&C^u*~ckm(#ov)TjjgYqrQZYG4g#jto(Y)MeriIpM z7J)BPbKKjCgCmO~OM%jW8jg{Emk;#d6uW$2mk<1xzz4K|BK%+=$%WOEL^{$}vpYSAx*O82H)w#n|l%Ier$^`lGzELOzue?F9+OM^z3SN)k^+T^v4ZBjt zEolE-AiPOh%hn*%!pmlYtZIe&1R3^(tJ?&N6%%~dFk#;`SW}Q6d;H}1V|h^#`ZzSf zFMj1Fsj!sx$?Dv2~#?bH8xQvdp%*V7bshQhXW1*igzTpsklrThR2k?I;*jFdw6OJ zS=}!*ZaUrXfT{ZkVW_390t0&Mrlq5u)>VO@x4;Zm>zmAKeT|$ptHi#52h~8qWp;|% zclHL$xxEuDgbU1fD)3D}nviwe`cWWnx!ST5iO2-Oy!ohh1X71ViY&jz5-Q2T__e7s zLg@yLz>R`hT!vJINGM)U6l?FibcIH>G}kLO1jy^r5P3PVx1IfVtc(C%5wh=b2Dt)JRwmVWV83JasS{o-VAEu#Lg=g@y>rVu857 ztWGjM-a-1MT@ttUIXrx^wY7zu&%pI>$$;0*DDP2 z?P&skUCy?S)2~=C?&(3!3Y?zr4Ol&!VDSGsP4KSg9cBg=zdazvOAi+LG~^Ux1ker1 z&a5w!va{p!$m-GN_-yhxm4ZH>5@bhICxsY8el;~)yXqw~TA%%Df2(eNnRL}Ry~`wQ(o}MTO`w#(73ols0~Jb% zJa@H8r~NI8^<~mUdBj>2=yyWgXJ0=o!*XG{T;XLA2*ysObwbJ>h223vH)y)~Fr#TN z{gX(KBi)w-uumE^T`xaQ$g!+1lcMR9$0_|Ir$`V|S5a#Fph3FDm6W<%hT6+%lZYrG zjzLXcTLYyc#*u20H3~?e$od#47Um8r)u;5WW^tudEq{b_9}_(ns_j4s zWT#JI7F*>Ny#jB;nd!ytybULZE%~*5)7rNiT077mtyy0trM04>wH4gi-5<<@0EAS2 z36j6i(^o3U4NfH(#eSu&ilp8mnoM0Lg`-8qjivG_(Dd^bjb>=#EgBH({7B#H{3)}} zv;D2J^<~m^J~1y`Oh`E%ABQ70rAIB{{Ax@G0=lY`#>&v9naW@Mx#P~^?(Mr5pnqIFpUN$~TQ#`QoT0cw02kU3dI$i1y zOzTSnW+YxWey__WVk}RVN$HPHV3RN+aNWDo;uv_gMxHINlCu_g0+F#AyaaI)8>v_wZ5T6np}t+dn8)qLB*OeArw>mZ$xj`be&om)qVS=ez&L`a z!%}&@ge+2Z)LYm@TBp_GYCiK~d9#e;y3)oVr;82B?p&KOeX!|Q zjjt@TuzTKOk(6oeKKAu)?o{sb5~5m@d#tzYF(wJs8mm_as7I8>ifGD@U6H%|TSK!p z)OPXjoo?!lWIoL8EVA%%o`wGnv@ddh9^M0%!XgUolR$tn;d zCad0b&2uC)DwVyEqw4N*YA_iMj-ge5>J3awl>)9aYxTB%XKtoSZ`G<=TkNeH6TITf zpB2!;XbyL zV!VBQqgfm;!@o;&8GM(_YMrcesOzlqw7EZ)O<0SRps^3{S+HF#Z``c6zrw2ZMB}Vu zJw%uL9_hoPv>|nrUPOGO;B^*U>PhV6_e%DtrmlaRiR(w`EJ$&;NAL6U!Y67eJ^@D zeqJ@3vH9ma#^cyY`^h*uoD1AJlL((;5r78J^4{X*R|hze{|px$ zhMhFgaH~t%W~fg1O@F{7<*uJd6PUZa8_|NPM}t_<5|XQwyA!r{SM?vKiHNU_;VOFQ zPFzb5YT}u_wdV}8@893-fe~a=ryS@|B+GEar;w6dfS_Ng<3hn55e(8{MUlCX|Cd3~ z{DX#vd|rOfJY=t&okd~#5fG=^Dly!dW#^tAASo8v^Sxd?W{SU^8bI;uYT@A*H6Eh5 zuyiJ|zWjOpWwdYH2hK`cLy&D2>g&ZOg2%MA07tYGzsHO)InrJ#%ab~WUT~mJ_uaj< z$^(21diW2arE=+E4`#2{p8s~&`L9ABN26Vep+9U0R}IdDd!yj`7DVbv>@<6lrDT3R z-@D+v^U;2^xF~Sq$=1Tco0|Bd@h8dF`Z6im{wY;=FbR=tTNR2tI~1%fT!x0bOlC`E zvh7_iPwlWxuRM)^jftLNa)EH`ePd}r3p?p>Bg{VCHXgTDA~Af zzp>O#nIVL#lO~+k)2=oHbj0tt`E~Yy4=6(9zxA^76hE5g6i1Wt8$E|k;!iw)#-AA$ z&v{q^n$Qc)`qG>kkprsBa!W`uVzap{U-%l>NXXEzUW3U3mqKL4F!$mhW;owAU``+d zM-4n0erhYtNn+YH@NDG;*eI2^z}TCYkbtV&&Oz5yVe=T@v8-o8;=15+WE^D1edDYOtda*$CL>mTuJR#SEOvDRcj-Fa>dGBAJ;VY|-=! zCM0i`DO!?vWcWrIqBurGm&l%q%=c7ru3iLFYG78lHAe}y0@xDOl|uo65l(HryvF6? z_|PPyZ4d{OwXNdY0_|T8Z4ZZ8L<-)&7y2?0Pxx`e6V%Ae6YlgagawQLeeC?BYFPcd zH0Qt1rWkA9?sLz4v8LS@siB#n&2uRdp*6i!yY)bqoSODMHh}iti<%xxw8<cbw zVX5y58DEDma1X_NekdrOSZiFtKlr;v2cB{4ihnvi^)DsCZCjLB6?aq{#Kp;_RD zzu6Lu%bB&Tp)Ht@a*KsV6W5@qL0N+6U#(HC8ytccP?idQU~spE7aKt(v6mS^2P<$V z8)LsMjbxNu8x17g!mJR_faqjLsLyy+4OCTSxzd3U3q;tX1*Hy!(;XE& zg`KL5#EZ=umna|CqbQePaaoKqTkxMNx+oPfg}6L8&wr7Z++u}YMGIs{ok7NTTO+}@7qE<&jO+A zl@E+(tt}IQL^#nZ@LeTimRZp861grS8UTO1MYXF z3GN>3M++qW>6v!)>Z^oPXmwxYp=FH})6g@A4QKDav+Mj*GK}Yq7G*k~ml*12`a)^U z2Ht0%!xqHua}R)tk$f=9^d=5>YuqHQJ+nz+WBz7eXqDEWUVL#MOIR{xkOiLm+yh85 zv_^ot;P;NfV@+2qBL8uk;AWFbTxJrT~q z>7kve{nzuo-j1D~O)&UUU#N}|681`s(SlMx@&W2NlRs)J=Y6P|3K0W;uMx%PU866A z#2oK8YZcf+iS^Kj6h|Pu%Myx==Bz;O@m2%W>Cphgesf=7#;m)~@)c$ak?JGgXjshO zXZh-U@RYKKXk&sfX5hj>2lnx$t<0G8k(Nlpq7x<=43bd;Ia-!BMvM{SRGR%;>z!`D zSiA2Ft)0tp_e#hVk?7x=ESS!JB-BYF{qh%#0YpuL4In-OPz{tG9ZvvO?_L7U@)rXGKh5$cWcDcMe1CX0hA&4uxP?w!09It%u3hn)fQspVaKDot@N8r zpzFBhH?((TV9+c=8=@0X)h~qz!lA0jyRcsmeVK@qe#wxM8ktFHfA>OJAo||M&OPcL z>|dri_lJ_vlLb%GFXCVv%(3B5(**w7oPlfO6MG>2chi(lSW>I6_ypQKliL2$JTJvQ zJ#fYIHjCW7yln>-a9GR1MTGGmw;~ixL&ZgFt;(uOHH^$x!lcd|Ow<)cf$CbNU9D}F z;s4{f4N>kba3gsQQ9+`t@@lY(v~hxG={KHI{%6A&4W(>6bgLETdnohN@$CEbQx6xGrswgWnfX^Q z&rHtl;Lln7IWaB%KE!|TY&jr>Vtp;Y3W|S=%i#1r{P|>7tVR0{!-y0rCPv$%yI5Un(2z};^06{M(s)0OFB6FipZ#7e8kKMfgqZ^lwsh zt>$kYxB#rC2|$FIC>q;0ihZXhtA4suy_Duu_livYc?-$btP{w{WmYu7 z1vq9sLoz(s1dOo47Cf43b>Wk_}}lfNKOK_$|t7Yb7gSA6@VgRL(BKRTkNs zYb`j+lD`z3vXI^ul0JgRq=gF&F>J=f$(>(8OZXM7P?^-#;#vU}RYmXYy?d@uzEr51 z?0Zz`24#OJNAzkLiT0X#cq2{9CzCP<+T>QbJo)0(k;9K3ntt@q)T2{Vk32ekD<%}(Ke@I1DrSE^O0(t?Pgv@plT6ma#9 zmlUy3FE^p!PzLN4#Ye28$Q=^8t_goI)GI`d2gU}KE?j}IMXprz1&L`;tMLCvW)8u* znNojQqd-n8G)3Am$Xu|~u!)AeutL%q{NGl|hq}QzF3Qa|^%qE4O@Jk@8n7BjO;$wG z40PhpLCXRtsNhl=xoRmZ7<9Fj3uM^BqQUP9UfSBA6r7TE4+E2m%SeFVaFvP9;tE%N zDqu!`6yh8{PU~r7Z5<0B`>C&QlT~oUCg@)KP4VqyjWtW}Bx{ce{&@)a-l8>9LY`46 zA=BCXI$hxp8&O>$nnm?MxRCVb4i@6$+j^G6J$Lp>UZaq%e`~z5HgfBGUB$mQ$}AV) zULan)I(*GS+|`;v>~dT;b4aZO{pl{k{b?@3sWe%v*$ZtAxm2f&zEMsPTs)DzY*KS>9kUEN&uTj>~~T$gDLuubB`lflFbb7e>4h%t!k3 z2aU}dQqF4VjfnQ)D@_U(m%8#}VZ)pbF&X#deJB}WDu$M{x#kaiZhtT|g+%A}j~dZL zfxx24r{0!(#9Gz2=>+>Dg>CRC%B$SH&Ap@$(cC44a+;)&q3Z&X$IimP{WKQnee7l` zIh6v2DMa)T9rYx3;_vKYr)u_|H@bh`j^CCh2)BosxU~}9?_$B)gCjrz z!hgF{|7@C5fBDKCH5L-C*^|$+M)kAiAweSEEFW6Zp zgjcXTiEGIh%T%?hwoJAya_518Nl8CjHC#th0@60(0h>q*g5p+X7|1F)vgg3$hD#{; zE}^;g5uQN8gP^74CJ2BOp}zuI43Uz^7zLpbjlyX>WXOGm%3ndaqOADhLlI=pvHlo_ zj9nCDPr2{K#)529Dr`XzEpyf>47~*rl`bpEj;P}gF=Lv7^qOPW1G=-+vxFT|S37o` z5HOilsxJbI8;N96R>O6sK#%6IOhn{%@SSf{lv-QvZ97Gf`Tri;aE>?@nZf?ILSH75 zqJGRsQEFtCqCWYyBR^~X-iDKlTe;32>>kBv`ESki_{{x((p)?xBZ9>H_84h}8U(mC7);4b5G zduOTq@M3YRQLfagNKm&3H>;(~8*@`TuW}iYK>FyStVsFQ4XdSb)49T7zp+FokYz&;Koz6H9_yFN%TNTJb4Nej($mjxn4Hgs#1AJrIERiq*m{w{?-ayrB_`Xa!Ru`4ey*}FC)vIf3 zDSNj8v?oF<;x6u9IYG6u`nN_a6ZrF?P7*Eaub5?3li+2Y>Khq)GM)av^`Ibr!L6yA z1$Wrk!Z$#YrJN1xd}vParYRI;p$Lg-XzZ^}%O5lJW9DJDw#i4# z&*(A#VOZJJ;v%}6_H;LGcQ@_tZaUoEG2AjEZpPo;Ot`xlbjMHx4~g`z0r$@^ zc*fx8_Bi5!7Zc(Y|JPuHf97dWTxV-^OQPLfinqa1=q<|pq6B_1q=W01_TO<0Cn-9x zVYZ-kAl`Ybh$r>_!x%cV;0o}_ zXtX!+4@4PC_M)y9g7(skDv72p7+yVRq?Uoc)nGB|zj=7nzxiPgkQVCqAnZpS82q=M z!N#WrV!xwr#9sIJJY%|uecC}RvsXs0WVkuWaFJ1mIj=^v_{l5;_2&$-Q|nKI&kOK_ z!!N-8f8s$#vIY1Vu>e2S_W~UJvS&Mrd2p!9(JY8FkLI?kg z{Q}SB7lHSGLJZNf;*(JSlCsG|{8Y~pRsl6C|LGl<->A8=)S=L$B0+X>DQ}r2l?;YU zP4>(?e}DMRdGYwqT#9F=!=b$@t zw#1NoDNOG&d1E0s|3PqI3NFHbrT4Y*)h$-@m&zB)TkzB^a8YjhHEDgZB>3Q%v7U@1 z8Prn9leLEcdh48Pc_|r>$oHJ4417 zK|ogACo(Q0u#A$s=L7irAVQs@M&a;MDd4V(J|UMF%FVn5&ty1T(}mHtyqYOiij*6g zAD#d|K@Iqfa@rxoT%!rEXz>NvE7=$%24IuW$l^+^aXGs|*32=aeS|G(WU&FWLUtn~ z9*H8vaIMk4K&Ur_N#VbUqN}ZQ(Li;tL=ZSeqYUQivq#b86gsH}GCiA$jWyh`S!0r; zE2d*GTJ>o(|Eq&tciqW+cf9Tk}oU-ZLM^-nNms0~=lj_HCQ)nR^}$LDGAN zTCTKz*X%b1CGWSd5x0gO(ln!XA{u!hiN>r7xk-3*cdU@}Mg5maurRH6#0}r+1h5wP{}R5B=Lk{`aSe{48&PBE_Uhp~sigPOsBWuhTtxopyT-8|u;PjMM9k)9Xx+UT551d)e#DJK9>0fl$@W zLJDfT$zOz&a*u(^?m%U-(G%{XE(XvYzREeH(E}PnnC@hY9%E8lcUV@`d)^Q1(!=Fl zdcgIZ;k+y={46--ke3v%|B=>q3^TH{)tAS|&(vUKP|0Y7yuUREUoMKO6z*gudj;+c zHjD6hTcH^e?7WdJ-OxGL zJsr$u$jPXwboO`mZgB2>i&O85;R!ow4fkMrO6cIfU1+}&`Z$ry{MlRYQg%``ILpjW z4vPmpx6aUAc@PtMHg1e!nUNR9f=y8Z!Pc8fl-wTKJKud-U3-ar1}Y-LlIdDbRK94{ zS;3Un3pWoOS6mRrHc~I&6Z5S_MzKwSy1xjKg)2!76jFcv6RxQy5vBd0p)@rmrnEg| zXf)KqKRwfqUlVHpvnH-IL@u|E5z%Y%$B7G>NZ9?ScZAu02EX0IoNoj8?8P~cPayE{ zClGb`69_rJPsDL8K7lmDpFob`PawhgKADlDc#Zx+UsiuDZt%DyJjkJi%;6hdNf;!} zkjFun?SB`kN`^}Vsd^{y33eZL+K}h+=MHNQZ?=VXF3-~q9fiOlPi0W4HQ02g%wWh& z01Zan4;0qnrXKXXCP<+t4gJO>@t6Y}6aBCnX3d{p^uX(|fmE&jQUgu%ZbBo0N#MQ1 zlfak$jt3H#`!+E?lZzs^){|>6(ZgErqLz zdz$H67LyWHC(D(-DWMNY&ygj5=mx4)rLsp>IAuO~2EWNf3tAMh5-k1IP=*NA|CFDh z%)<)@Lt){4Qp;jb4hf%iJPq9yX!^MjO}i)ZUCu3*ax74CUyrvg$(#!fkm!8=>W^w7 zQ4{3z`Cgsiu4@@ByhOE%1$n;%GLwUX*Z0Z=c0EmC=Q+8{FjQ#aB!$jmQ72O}>!39K z+eOlAX(H(+y}|S#9rMgH<#^f6B0ldSH3ylP6-FUaQ%X~Ho9QwWlBE$O#qSmCGNSc? zh;FQfi);q?zCDCZcLNV3Qh!gV%S2oD56qTTQ{pYV--80PRKB+!pyC(J8s03LbJFgb zA7Qd|&2wbbbQK9afd6Ca=Eq(G(R6G}C&~A_=0~5A`zm2wx^{c#M5_G~Gnv!F7MGw}b*khzH$tY2laa zBDhpbDUMYS%0bCRMvQ6t57V6fsWkVG1$essdQcWkW6Jo9R=2u)+L~kdjb{{hVG@(v zmQGLI&nGUGUi;(#cHv%BdOQ{OWZ^EQ4&zWB8N+;MnizUlPsXd4rPI5O2ab4=-&!p1 zDf<9U*^?sr_B_&8)N9D3ieMymJS;Yh<0<6<6N$4L`E{Ijt>t{VHhCI(BXD7I3I%Q` znrU*iwlcZZTAqaYXXY&Y29Fh*g~?KLv*J9-(-S@+E4A7MI@-nMpq0V4t6qhfiQlX9 z#o}gp0-3wYxPUBF;04=gEUlq(IWp@aGwmsne65Z~I1RB`8^sEh!iG!%8|o!lZDSP` zGM3vgsn#yg@hE4tf~39XWZWx}Ws#|Vgayil$i?=F?MAOUX7+nt8H(_|#ZZSsL-jsk z;1l~H90D-B?g^XW^|3v!(DyCb5;aA5jqHwb`sS zD6tJ?!iLOG$*{%7#0fl99~wy;>$LUgIS=sRah!I(6#QsZ!6dxu_J&^b^v~7uxXDs1yqg$69{90q*mzf>TJY z<%`*x^Ze4q4=kKsPA*-%xSX6nyM%kA!%Li3vNr6TIs5DjOAE=f^VZYJGYseqy*=#A z&n?f%-)YI{h;3uf_bG#ZNkP(u!6a*O0-y_4jswG-4Kn z6c(g>wH7nKwR>A49q_LD#c{Cb1A~#TJV^w^z`bNXBj2cI>l9?5h$|p{l<9SxI8AK@ z3}Uqaflr-5$j-+p!>0j21DFIzFA9}nJwdKMUfb1Ammo_(ogaMaKPl4lhrm($5M0AALHG+CS6B&=>vRQ(Op0e!#== z=^6f|1y&BoXREWvp7J9I8zj4x1Rb+yQpH*p+&StlY}HXSsZ2hFjJohZDvzoLBDFmp z!ZH~>I*JFbxYxnwedO*)^O+hy9CvGF0HcYzpb8F3cfe-3G|EZIWER&1bWFBN_d}3@ ztXPJzLyiJ25Qf*0q)hq?5^_K;k1B|xSHd-O>;Mp0X?wd zLxTD$rDAa)e7amh0um& zqK@0G-B6u|?N~9jv6{gN$4R$-xt>sLoSikF695>CUyVC+8pX>%?W*%3>-&fBebM^9 zi0@LUt*1)N@tdqPY5%AM|0=O>5xyU+tdzb*veD~1TX9zKZmE)ZHCrv$VzDCT`dobc zL_F@STs}EIMRG9BAE)W#9DkhQk7Hx8nZ(rOvBZ>UdrUo@n6@7uw;$6$Eaz~gr<8|S zR^ZaAlXWf%-HCP1ndh2W7evV{RuOu1Vk>da&k3GpjF#gu{v`^)R zukPkX@c-|*N}is=4jv-J{fPU33g53uIDRAwf7kzh^H4flEGFyQmF#ZwHu8ut$SneC zGE8y8srWe%Wlm9>-|-ThwcbNUk!Ydy_)C?To2G+CvTgF6QEkjaBcVNWwn4>yE#2{`LEsS3rK$hmv=c{p0(0 zL}}aMXZ83hlGqEn7%@vVs?De&2Mk7c7oSD8<^J*%E`Hka45CVwx-*NES9VGt`z{e{nD$V{c&4Ps+7NBcy$9<#Y%|w6PFw7$i^%jw?lbMS?9@c# zTlfM=;!xr9{Rkv`B~N|4Glt*I6OuN&BB=!Cxy(=Q3!}h(T!t@jV@es09Wa;N^C=fS zt|qKc#0QsDfG}$3wT>}jkBy#~j{+@A%|rtaBk)Dv2za&B10J$8Ba}93_p=^nmdiiz zby(GmD6$jB>nOWR4 z^V$<8*$kzKBzr=^@bb}lCbTkJ`lXJlV&x$m6qAP#{>O3^eogipDz}*m?AHP39M?oQ z)Dhfo<&deBi`P};kQ~S!tMJoun3q&IEu|4eNp@6IMLZUiskrz6Hx@#o;9y2D z`eqhRB+^q>;+PRPCh8-YI#4+{uh$wyB-x?1*{xK%F5igjpeaod^2=zy?1+`=8>SNZ zn~U4p$Qkd4OV%vx?>)@R2QldFGhgetaw%>4yH|3K`hfJLZdU(D%ZfsR)y->f{#aWK zeJfzob(n5N4r{eQ@MrJvJRarnoQrS{SbUBYY*&WeX-z)GIL|XL1J6duz&rG0*PS%Y zvjEh)uQk?GYPZ(>wsSh>u`Kw$O?>vsS~ZovF}-u6v?FQ^pn{W1I7xTIVi2tfi(8^( zLHy+z&zscp%8e`hg5AsZ&V>C{!`V?_mivE4yn1EF81#QFY16{N2Hh$oFyuy~>byfa z4UCXj8;xabyu~pItzH17VbojJ_}hb^ zTk}8ocg%*WZ|-@m<$-S9-0*?yX}NlBTC-1~#LLwbFY_>HdS;0pS>*DU%Y4u81~8<% zdazJ8Bkwgu|8WoMp~p!cl#sLc>E7XOHStXz&u=iS^}zF#-VHycKenP}(hRIMrJv&* zDk?1^-Q_JKF$Kzz*77irQB4e1t7lJcf+vxX5hv-5Qo4?mklND_9YD>CW^^zyIexsK zsV#DPd|PK8PM$@y7($hbHNLSy0aco34rxloBzNSr$NwCsObek-(DkG9!(t2VuFysZw*r0{CH zOWm~HOuyEUF-C3*Y%kycAiU;y{ga)go(uuRPoQQJ#6884%q&m{YOBMEhmwR?;StTZ zna9BZe32FSBxznyBjqL1+g`Pg$2OONz+!+kK>1^5P*^#Ij*8Gsjk-`XeLHE@3{Sbg~cLT`cl~D=SI(gT!rNVs_-=?Tn9ncCGLPcqp=z) zr&1HdoTVGE_F%NEW>E(g_hm?7^V>4|u$aG=E#`COGH%eQBCgz)-KTtDOEGZ3+u%Y) z-N@FVJ$=M5?EM1SNOtR^ni2@X#=Qz0L9ylMkTs4B16eng$>L%~1|j&A z_Z;>$ymi?)R*Y&sMcaP?kf6!m$RtlWG8`YbDo_$aUa=qmDY*4Op-TZo)YAHuy;wvc zsza1ZI15Xa9B=%NT=VMHhCEP9xMF(@kW`}_t_Z9qY4#_U@}L$G_y8jEjp3dag$ZPl z2OT*OXqnTcEReZ@e2r*;3!5@MnQHFR86OtDZQiD%9WFUlaXTT4OhM$;P^%+)!`0Iw zYE}R#apq^On>?tSS%+6b9t$%PbYa^X2`;Bj{M19Bcx`qzs_3ki(58BViT^!5wd2)U zZwrttWvmANru*Po@Pj_!JuPi}k-1 zCF}2FJK5Ha7Eu8|1|Z)vW6wmRbu9#=f*6LX`6w0vi>7Gcg`wz7G%J!tQ&bScP<5IX zt&j_(MN%Z-!q9W4wOCttc^(JA?@QI(dujmhSG(mjL0Y~KYeKK=&3@kuAiMka!a_U9 z`fYwM%%;|Ot``5VtO73vhwMI>zA)up6}}1?dPQ#Gz)%JD9_tK6bW3I(mz~$iok9^| z0*DO#(!K~X=sIF%iAVRgU|!re>|UAC@ll=QrDl9UdNRfCG^UtBf=#hU$xqakrxu+( zzR;D}`vZqbS)^@#)so9L=|Yr@c+8XFs*8YVJeqyI7QmG5)5jV)Gkv_f^WSl9T#*+< zYxQkx&aPl?ByV_vXEV}Qd0ac{G%~!K(7WVayrl?S%c{x-l^)o&B zjAQuc18DGy`d1twtyi!<)brT22da^s8~S(Qn=0(!vbFEcfJPioc3C#;7S`r*dLdFx zRl_2um;B0=pWyVG;8daVAs1U>Um2xURlzMY;=|S1DdYS4n?bA=9G(i_5`HSod?`ST ztf_FX3d627ipcVlFDZF%{Xyea_@9JOH~SXqQwP6EdHl`9^a}OoLkRVRkm?zZ!6fYs z@vSvB;EL6um?Ue39T6+Z0hy6XZh+kWwV_83Ka1uMhZoJIi2z}+MDty6mF77bkRvEn zOeE#LfswD|v`Z_Eb+b!DDCwWp&mV!xnBq`bJZIG17^wG;g_f=S_ytP*9Rrn2(nW4fL z`WBtnJ`sS)H-_k#Mft5NI>1m55sS|6h8LZWeIa1LEYaC-bP|2k{TKed5Rre|w}^b= zzXbH?6_I7nE#8$KC?A4UO1%tZ}!wYfGyoU0v3YpEXu5Bj)*1I z#pylW$?42!wI8`bdy@{+XnV)3reCMtG;etpjbdy=?wQ}n>X9~#G_z>su3Pu*!{SKE z|3wHlr{)*YrQ~n*<|@Pd5%)zA3l)@9-{*NzIsZASKFuMmQi#+>S*{yoLw> zO3cP_Us5?xT4J|GAp_3H3E45d$p)(_yqa>Wa5Le=AE)vZ_Fl`^8+>q%EEMqFrb^o| zIw;GcuNTCbO=KuDBgLJiiVnj~qa8f80t4!U;@6NuGy|54AT=Dr^I1D7CQ1Ur9dO8j zQ)?g);d;J?CPkSg8lx#ICe#oQa1HiP5eahgYn11Z&i0w!VtJiI0}vCjj+0A{g9BuM zql~DkEZK(0KV3t%!L3|=6)qJE`+Pl56z1~joCAg-E<+_KY1ELBlL8@->64Zg2%a2tdRWWF;4q*)pYUmT z@1oM0mcH#IR0~W$uXq*GGd@&$T5P`D)GA1k!i&ve5s0mHg2gWXFI}1NbSE{wveQwY zD6WKp;bqgMDA_dJUEf)(>!o#9cHS=ar(%u%?j>_AN-}R5H&hh9_g3D1DRHr4S8nld z0Rr0GWh;FHwdsA{l?y1|aYAu`KYOYF+fh=V=xSWm}$~rePGdCZZjAS1c3XVv@4WyH4zgHv^Qg#Rly1#5 zuX%I+U;ARGOkCJbxNb=p9u?k8g4k;JclVlD(1SMn%`ajvMY~#DrJ@+{a%pVrq&r`B zx62J|l5qKow@L`4-Kfy*ce$m;Hks>R^X4-~n<-BHj{TTQexH>&(Btf6#}1V5aLG_3 za=FQEt=gN3&6wNG4KVV4l*T~sxLu2KLTltbptMyr?uGW$2)|F)bv^6cLGJ*)Z60jA zC3mZ`9=*Ebu$L2Rw)r1xkWK$D7@-G!J#^j(xM%quU^}9{efo5}}uc4H)x!q8{r$@mx!qw{=0nF*H7A!Oa)xxilvuUkN&KsJc za(0rNN#?7mW^u7nIE|n@a8VQKE0NR9>y9GD03Q+!rxNQKEuN3(0#d zbO=ZHRN(a?_v>0JO^&PbCb`ncnODrxHBP(&6pVKat>KqaZLAkXYjcIH zjV1)`S>KchT6qK7#PvR<>wp>g@8hKa%(_p3|6-JG;1sy^xbq}00p#(!<+wFU_<|mj zJ@6u^&674H^w0{aJMLfol=hzm;{LApI`qFs@V^VroU`ma=Omp|&UxpebJ{s)sCuuB z%Xg^AK`DPhkhmCNcB_cNJ!0rAY#j4pC-PqRj$21yHW7ZmR4}O* z3~tR^F7!VJc8_xgR^i`17Lv|o1FU(=ayZ1Evtd0K4p!2Jl?(^#lnv`tI9TUxSm(pR zx@g0?7!H<|f~fF$N}FO&&3=XdoQ)VXNh_@G+Uw7{XfJ5qr;LuWG;d*DXqNxhgOUnr z^OTJ|x)>_>TXjc%|L;exdX>Vwe*uGI?M{MJQc#-!Woeq(>cOv&f?GulEEFWd zgau|P;KKhM7MP`h3;!l8umBZ2xM)}^xbWw}m~~fU2fC=STTeJsMhBm;6(>%>h6>-` zgRTl{Gi4wZ_KPwq{$|~gKeB(QV;tB3=C9JhKiLCaL2Uw*rfJqmF$K4Z7+iqbH2t>M zbEb_=K}TSAL%$9V1{a`%O~Zb13(&!X0}If>g98iD!Gi+}(7}TP3(&!X0}If>g98iD z!Gi+}(7}TP3(&!SVBOW&{XuHX8}*y<-zm6)_^caOsAj7j-!iwuk9ItlyUgW+Ilfrn z34xqL!g9nq#$R4G0gYdu&a{Vzk4$1`LdP=>$2+eEC&~8*H)KB3K*}KR3B#3arIs(2 zO9%uOH3VHQ4Xv`^=oRp=7lv8Hn*Lv~Ha-cOY&+f?CUL(#aJql)2?l?gU%gD5^L5bq zkU@H;XP(z=QV8-P!~MaO>mHkJ8Uzs9UF0n!jN{5H@>OpTuNolVtr*|+7n3EwbGy!A z@NY)fn$J8PJ`{dBoZ8gmY*=iv<&>rx56K+hIiwrjfwq^?t(N_2&TK&}kCRsytyF6!ABc#7;ObXv62IwPw z4^Z_#1oZ74pzYQHvO?mu%9^zmS+v9WCSssj%!BBSClqUVtWu1W{@obH0ssnWSV!n2_P zxHnZ&Js_#6IX`>2Gm_SN*sUeB$|rq7>s=p%ikHdr+n}Gk&^@h;A3Lc>NTJ3n zV_qd}yZL<9R5ZViodyg=_Z8e?R^amT;(4lSU*mL)$idsFa_)b+H_|RdgwJ$qc?6i? zas*1OvQ&zD97xQ-r_82q33WA4V%Dd-1p8U>aDxrv=%SpSu8SR>RuU-E(O zN%#Dq5gLUW>mCH{cN;&8XT7xUO8h`|fxa3g(MMWE&>}ME*8rk|PE8bRRrCFm0I|DLv_{cP{lYf5%SP!RjEKR=BW+3zqhzFV^a{S7IyUuA1QEZZ%zoj+kA9 zts1{Pbyd#jQboK++5teup()~upAkdV=2bh zg&Ue_Vj6Xne>PU%_*a@!p=5LAbws9Lz%}A*?ege?$Xz>n)rHiFB?Sx5vlI)CU8tK9 z{1NO>f!`)2I?|fyO)2B^FsQz_Z|GJRmY=G5`8x*d-b7g!|eZP=>aYw#Fuw#m{xzVfR zuKKbdgWTy^s#wXTW~UNQjJw6n*_m<2wf^G{XGiBYvNKcDj|)sXfK}XUh3ckfM>j?T zfuG6O1>j#a z_p7;{bZt|sjRAD+m%pYp$VVj{KN6AZ+COmfMtZmvV;tP9%$L)J#%3oEkFIxQIG&5}S`VPTtW z#(IpkKTtJ`23Y<`P^YDTf{yj?!-Vd(9X>_1^jcDW>8_B>}YI=lN<0D7vm?{ z?vnoIX?KDgB*+IElMV4ewS-o8XX5T;YYV(WryTN?cwDqplle{PSq}lP^6z+Vz*YY3 zcyVBl$NuwLL*FOi+|*Tmh)<`nkd_-oa(T&R6x%72bgyF@&9uS^!gr`}Oq1X1-snh8 z(;QA(AFNbiE5snLdY>+KMR8?-H#;@ar11DZ3p)I)@5#!W*9;iI${YS+&5Jn+$B#s$ zR^EHOrcNq@GY^`ThXPO{rp*vM^pSX zR4DoYbT)eA!aeiLnxzT>HhK;j>P%18!vKW4YhJ6faI(Bfnim{Vs#ob#p?OQf%~V5S z3N#Qp*;dJ3bv&V(jE&92Crgea<_SS>J5k=W{`FY!ENo)L{S%<4S@=;cDg(^IQ~yqj z%y&vSek3BDg^!RHR(>toPl@Zk*KxtVT15gjqlM$x|sJ!{6ZoE`;3I+M z)>-%8gS0gk_}^osOo-XaC*ijB)WD<&*?wwZit3z>UFr=H9vdir$yFu$D}#w@lco%& zsHwa~b#`>FEfw3fkyR~so3dMkrzDv`F9 z!F*!C@v@|QU=WbHugU4Fw))o{UK+TMNVi?Vdyv$%?!85i3(yEs!oMhI2J`w zKnp|7!<}`Xg=x2z0a)~tOZNklC?k{BWT+RZAg9BS$XWC&R7DL#gQwu{Q5?UiWXAJ=Pa^BQx?-&qsnoO z4wug!r&AK?2c@Fjw7Rz#Xl|^Gg4O0cXL_N-(nU$g$6Hr_Xl#+BCbt?0CIvPPxYl1L@&u* zR`-uYN$A^y+kthjyr?%vM!fX-KzD-wxZpD;P63Cyu zr2KT0q#P7e&q7>12|%;6cv*_gQ#_@~RaIK&UU$W1%^0J&JU3ca9PD-+v3c{{f)ugR z@maeTq_2#aET6|Wl(>u-M@rt<-bv)>dhN=M)(RADudgYX%Niy7 zmd%w*q+Zd9tZQcf)(xb9d}T*pmfGV=t{%6z;+;Rgh56(6RM_s4=0U7o)l_BoqxjU= zSj|><7b&QXw)c-BHz8Kk*Zz*)-W$QRR@CGspCHou7Zj2&23wS zvesMJC=y6kTq2NnCTN+wucaevCI|e8N&SaD_2E^fZ?gqY;;D2|;BQ{u{(B$ro)*TB z{Ugmyg&Z%8L$n)g!yk*xj?KCfY?gSrewYVd(*D0FNqak4o-L!QJ8TC39zaZYC2Dau zP*yk3b}0&t+ay#96qX;XyRM~`rM(xVj7MBV^N5nHuryh!NQS#tnl^mt-WEK%{{vo^ zgjd3QtV;c`4@ggv{Ba}63JI3vM|)zOCDT2=(3RK++b@l3R(EQm8P8Vlbd)T)+bayZ zr>5_R01~^)yfqF+<~@c(^RHU+FR&vyNNzp37icO(Tzyv<1QAv;HC9WO zCHn)A?2snGC*IJDm=QV4BxtS0E7zOWA2e=-D>UPV3Lolc;&Bl8>wg)5jdue4r?3;C zhf2LdvPHdKTPQ`{wl$RRzHw8X zrIp6IhmyZreb0_Jo(&+uJ3D@Kz}exyts7ML7a4kj>i(`$m38wzGXGfL#C}%pAx&bR zw>p}799jDQhw#$(huk1|PU-d?f_hdeM z-5p7BA#w!B2n{!>d<953geVK@Q|46nT*zLf3)z&TL2c*Uk=0}#P@oD82$P}Vk**|J zhGP$nv7r2bZ_0QrUnxlY_kKm65;aL#-#f`>P+-EwCFs|pB8 zM_T8ZA}Q!6FIoROO0t4Vz$nR8j{5G~982s2mtud(b7^tYfd`qvWKvNGEuc-eWBNcMvvj6|ky3-7uwW7AIiGJTX7M6SHA1k+Y zM}^J8a^x!kJ$i-3Teh4HCd(LF_18rAHwdzw%n;uk;NF3G7q}8~V21B>I~v;EYNEF? z<>uE!pYRET*Rb#u^Jzs->Edq9yqtdA*JVffQvpabGn%&wWi}XkDsry-@3|(r>Z5nn zZXsx9rFV|>TBW>|(Q+HXc|yImp0_yn>Hsy-L-wY5%j)f7_*r!7gL8@KDMFM#$62Ju zaoKqgj3$f-nkJnGZ6o}kt^Py+BdkH&R{J|1Hri;JidINFfUp{_Zmx=xdlPE02Y1Bf z;&%R>0qzES_G-_`^sys3-s+8r+dXt@@zlU5m?b<65D3_gPr$s&R-SVdlmP4R-qHF_ zACR8ZXZ2s`@li;y`uwK0T(yYq(5Nd>o}I^Y!5b1h_LBHRQIdFouYsE$n({$c)DH>~ zQp2IXdQttUC{ev7ZlQ@U0VsCgV^~9N^ru%ReGhaJPbw&*D;#8TZFUM-s4GZA4WbLz zcEr;>KP};@KJl=f!t4$vPs%mQb}hrr*wX6>CiZr;CBL@gC>Fov@bZrJuVSo2nuxFc znVyGca4bE!^%4^Q)hR;@Ih+s%O=UG@$-V$R*+8O$Dy1Prk|}l9JWzMy@Hz(lha^?F zIK)Xu@dS6Evs>9}I$z5gHF+UY3@mw$7%=9>7kzVYAa(i&J|I1*(~mv$h_dt*60A;# zDN3LVqqR&mdubg&W_J}}p=YGbs{+;>F=1bxfgQrSAg+~xHSOi3etAZiEUwT4S0H2c zJOW_ktC;=j30$J)2-DatU$5>obJ?)@^Y9b)6Q}h^8&+Bq_AZz%jUorf=5-99_o2qA zaOnI#vNLa|fBM>hzg`Hy!aJZJ3m#BmY4)U#g)){~4QZnCi$%}kP?U?qhF%=-;s%~p zDTgXa5iKEwokz4SmlfvMmoDQsT{tay5fIvXydiyM&nRRB;@-9GR*qhcs3U)NRf3QcC9Tu>tzC zwDmx*L$)9F$+kx)7yiBr$rtB|w``l=_AA4rod zzIt$2nIsAYo(cDfnPBnDqA1|Ut|Z89sVGSXMWeraiKs?N1h#BApS2K|4|+v*TJ10- zveO0>%lC()MD|^sB~P=d*`;9saox3wB?e}id9}(<3>YavR^k+G?pgx ztsBPRh#>?_5SVG&HCmfSgwZRzgRKZp0MIuKzItR?rhXJM71E~IpU`vK45pJ0W6}im z2c7jULS`NO2EtemAq&wjg%_gXZwVMsONc&rhO&sb$YGAuikvMs*3N8-lbM+*mqzQZ zDoen*(UB2K#>PMLi0eA@S>!9R1IC~4zl$$_`ww;K#HjD)V*ju^p-(=yf9{(!xix+B z^7i+xR^7?>>8*|a?j5!-`XIDl!J7x+^M+GdH0mt^__I z2~1Kp<2EF=u2+W4f#3HM``^P8yYweriJeo#Zi86A()DRv&#=n%*^nIpBrgN*xe$86 zTz%(JWdxfA(<{!iiUBvaw!~qaZ{-m(jrFI~n1_VoCH&pt39r7tE8*`ogl`)sy;*Su z-SFRaqB|TAFWHmf$$o7ufNX1RJB0O0)ag-XlJ!JmQJGD4juN&e)=?Fso|OFh8UW2GyT)VAi7QIM+Su2st$?rIs4&8t~TCDACs=aR`%ZUpTyE6)C&^a4EodbtTONOL?Ht!EPfEFR4LG?|A zWzyasC`cA|Z5jmh>2VnNa6ByQ!RuEa}5ncbjj+udZ-t>(nIpc?3Vd=jtwGE$4W%2AW)(xegf?PoM+| zpGD$me?b;(B;WZZ4k^G+6N85H|0m6+p~64yvJ4V= zloas`W*&W(U$NG?U&T6iDhrHpu7z~%K~@kR}+6jowB>%fXD z(4eHfDAvi?VjY+N;@l65Hl4eRlCjQZ5+bB!7K7jttRjXlIt+Z;pI8E$ws@fJ2b~xo>p{ z(!Bv~=A#fZi@0du^-Yhpmdu5gd5fF~fQ6yuOf<94A}A_|gHv@(J18INsuQ$5{9ya0 z7IB?g0pUOGVTuH_A+IO&lhu!U5&+Yzs%rgI3uW$f^K=9nju+e+tVq9MRwOmA*|Iw* zRG^Knd@}f$Jw~_xWvXD{1G7b()htVI@PY zWvtg#G8Hx{pk}+=5UY6&i!%znk{yCvcnHZvKmj5&Q&!?NlD*RWd zxj+?G%u0GD8oDf@iV7km?u1WZg4WD{Hm4(CvPDoN;QFS8%DgN)yxgkDv(3#hQ~ZAyhAgvL|ai3(4Un@1K~G(7IAQ z&-#RWrctGNy&*J4)~VrG^%~9Wu`%{idHadg%g|7M9~aJ9b9uP-P{*#=KuJ3{<>a<2 z*{TlhM$A(&wXvE>jZlF~IKc!MM2HumTVRm;TwlYaF*@_ZnmoO<#>nM-lIWG`!*&_A zQIRJWy5L^>F%Efc6QJdlp~5S^K6;Y&-!;;%aFn4?{t>o_zOQBY$@`+-@8o?6XmdIO z$+2*Y1YF_S>b>FhF^2`gC6`m${MK&9QfNR!C;ZyEF_S zuKUVhA!SyE!uFAU`_%pj1S9;o@U5+nK6968wes0J_w76Co-S`}z*2;UFIVeW4ftQT ztUn>wXl0vxAEk==Q7q*2tdddBuw+oB;bML@QQ4+X{ByD1>+iK|Dekgelv1DJcrlXOdhj2u9~jlIwPji(gSxr(A5*v*Kq56{25alNieh zl`dvu7v+nQGn{S<;VcGrC|m3`=lP|JA6PiOoLstiaXC4Eb_r*8!%KV@qBiWDIs5Dj zOAE=f^VZYJGYseqoI+?ZKes$5f1d|XemXhqIQccM*jlb7DzJ|e*{wY4tH=DE$s&Vj zwQ&vZtk&P#0n&(B5K`C3C02`>-`c$`kq&rQ{o=s4w}t^ot!I-&04lw&C-WI^QA9;t z0q#lD>o^$8Z3PTswE%%nok7UX$7y(|r2wF7PkT_{y3-h1JdCxXi89^8iArgG7|3u~ z((EX{p{O-XA@b&7ILno6DQ1uxUMOEmU>i9cA0Mq!!=vsJnM|;OT>K;?2Eyv14NDHo zk>1=;GXe%v2QpXR#BpA=n%Wi)PIyJxPs*=Qw&oC2R2|PGs^}*PurYz#xLd-u ztTr1PB_B83tE1!YC>Md+9*xK8HB?0>AloR4&H(D=q%4yY^ZYt4D%R9yM{&FfHwDNN zo#JmIQ!Q6YDb)5>^`!vI^ESfY1lIDo!saSipN@^ijwhxjk0qwWZ!^|! z+<4l0fW|X+<0<-WnstG&lnF!IY8$PogJx+?QT!m3L)AWnEmuJG5EEm?ov$d*VhXd=xGLPz9?9A>w~hbt#sUAlyWu zHX2uo97vAy5M%DsG^^B;tUqWt{RjH$I*-ph67~%wWhyc4Bt75IM^Z?n5Ilw$OBJcg zylfdHdg>MZdaD3(6DflH84~a&i6D0m1@Kl8L4!|#uym5KF26+~f(kb+fQ7wv_MtZS0ssl*dF3*DTZ8RycTihbO} z>?ouW?gt#{;fgM;u3e$CPHlzlre{YtMgzNmYVQh_IX-$d818y7+$CN_lzT(WY{|o7 zc0F6lfc4t4A>#w92c38iqX=ZNo}fElF>#qa_rRVRzh}k=?Ix75F6!mdCPHli)GvEx z{GJ&vRWMW#YhbOBvDogJ@q1=G2db)dx81+WZrL;AWr8n0LQT1QW_;D&9Vydv&y4TB zlT^IgGvl?N3l_H6B>Z(VNeURtCJ}4#XG!z;d|XXM*h)}$53B>UXPCIo{xIp(S|V=X>FSX zkP>u$r~0#whVSyQU6a@JJ`aH3{bwx2*E~(f)>$6BG3T>faqz2Js>o9 zBUOYeAbUN}Q3?nTsBco(F`76!Y?r*XQKGJgC zZroi>)$(awHUuwNmpfj&ctqYJ%H2hwi%qAR<$HXH(BD0G3ubqRiyq^bDn)tKgW?0{ z#SxmjTt)Z~U1Xqnyo`$G?EX_)x4)P4kQct0=FL#yBL49&eLl6!*E;?mGsx%p6{DtB zKDmG2zO$ULtVHLFMN%2q8i;ERMI5F>OUiq3U}ZvqY-$e=OqIryT&5UWKu zH1HCIe(57R$RX=H3IT?Sb)fl5Wh}tJEe-5&yCPqq1jkjEGEi+a5Xr^Z!a3O}QfwBe z=D;q#IHAyYxD5?lEwwEupeJ-o3`nQxd<21A)dnmAj$}olQ;umXQ#hL=vVq}5x1I$X zM`k9EjS#;^j!&K#5$)wcE9g-@GFGg`sYs-Q>zDOP8H}e2p0#vCq-5s4Kp}IU?-t2L z4!&3k7}0dJ^bn&m4u`AyUWX}Do`5sD2m;2&R`aRaggPh#i-AriTV7jBxX;mowV8E_ z2*<5c$XzX4yei-CV+*~YUjusc#;e_B zA9ySM>3yj6W7GQ_u(y2U(%F=iF}nZ3s7*CyY|4)GqJu1Ydut!SfgmGSVPb7n9%!D4 zM&VgFM+GqqRqqX>e=SWwmUiBX|3qH`F86Bf z1pGEys4Qf~_wFcSY11guXD^Ds7$u7D>1_!$C^UP0BY@1F=ix#mE4GDnL;%CkB(A%3 zupcaRySEL%#H;7;_822doP1BUU^RU|0Q{Hw60 zvW+%SvNzsaBx(PZWV;)urF$X8XXhz`HZZbTMI8JMj76YG>7|+2c9IOQaTHUfvw|qZyQA2(u3S|l^3F<^&KAP?58 zaMf7q)DDzJ@_$yd5LmYk>O+pbeSS586R`=h94GRn z3ATN3I*Oz6%mf*N{EIo0w%!wdGQ0=F&>a4;2}f2ALBxB$6L52iD~_i-pGUtNC!y{s z93a9IfKyIRrblp9N)K9_aN#5|cc}1Wmq8}i2_+NU|6_VsPVg(%sdH9l1o0v0 ziX&G#C*H)8P)6wyvWw{$nlI+img9xfXIl@Xm$ICywp`tIh>JRlUp2ddxGhS)i!`;w z5oQ6w=Mxj;AW`SO7zl3ylCd#dtHiLQxcjuCXbK1YQ`59v@Kg8JQduLIBpr3TSs=V2LiRdMzpNhDZ-9y*+8vHmxjb zP43u6(^dEf1b?XT_xqV)oTl$iGrbIsh1@}tZ$)Hfi533%HvcoTh&PX(UOIc>r29NV zP}8YO9=1igN*QU9M-N&45HW=qEk#mKZ8d2I`GZpkC!k&E>fN%_Bu0Iv+krqMlSm$Li2Gbg5| zreyL22urrE*ccr^!|Ui^LKuIu?*Tb%2ISW@TYDN1>&;#V0F&_AR!|s7WfV+R(H+Szf-2HNY+_m29mAmIoKQ{;Fp3bGJ zWE(HRwncP$V`F1m8^dl4sZ3m)ykWP51RSenRJf=iQ3+Z;?M}IzHGET-KZk)7DNHj8 zk%Gm@e68=1IbufUulgGq>&;$A=JCagD<|E>Jf)*@%UjSul)*g>JtT6YVgrG!Fhbs^ zr(EP-Eka+ZYx~VXUnhQl{7XNukN?}*jpYsu3<^BH|C_~ns&*qCl>{$zZ#O(XB474< z2J9_}rR=RYd*yL#35($z1R1HO5YE1_3gIV(UoLM5#%4Bg9|&7FrObPO!N_gz?|@+w zeKYI3nvS!^JsFB0-M@+6b!@GP+-96g}yocPQ&R_{c+lQvsX?(HGdX*`K2t?g9ld;OkSrPM(lQ2BM*f1 z!(`#qC~t4joi)-8II>i-8UljhatI*c@zuV0{Gj3Shx+5O^=7X;PAtvMpLEZnVAB>& zd%7Qhc>UEpoYrIt!z#`oIW^ze+C)Az#n^sdGu-U=O|Y=hHwzy!EUfm&LhH?*S-3oh zRq9NmSj24va)4D}I%liIKP3F&Z3P)*FI<|RhJCDcK%&prEEe|pCRq60eY5ak!@}?C zkA>Eoy|OTK>ZChD+=-pKhQOSu#N)@}FV7$;@g}(g-H}t*M%dkmO(-$f>`N|JvSqs4 z3elZggOz|QHgg-O_f8*>@R-!WRnMj_2IK5j^kz8=_$zqyzxsIOjby&B1wAzv0)EkB z#peq@?E~D=0w~BRXV>swA;!+6I}}qa3e2;vWb8k6P2o)+y@>vDl!%^+V8k?f%_yJ+ zFsHj`#2SZB^fwQzH+!82$If3`o?B+;1c@%0WRPs!eue&^%SlM)i0|ZDs+23Tl8_qC zf!;L?9_V#&^pEm?3JTWTv(WY@uYhJ$seH-^Z|v>V9VilA749$ zl6GmL>xFLM_YG$U1_AE=*S@(sVz~Pk{c+cN({Q)r1Szb;`R4U88PA+~v-xgFEXqmI zsosP<)f4BB@`{~x+DJ{Yzixj3*mA&V4(xh4V0_>?>r@euZ|IFGc&JI#?q@xkg`-RC zzI+7#blH^gapHT-a6HVf*q%rV@~dM9ODmL!dsate!o!ctstU2Ay|gu0P75q2>7TwT z(T;lY9XN8l`L|>scR+G$L>E$q!j?qVE}~yWZfraKH44GYr?o7uY$AddXRNi7=FH9S zNVAn!^V3~MzHPL?GkP%8S&r8b=ol?<MrK1EL6Vp(Qj zmK!ZKcjj>-e+j429LVoKe`bj#Gb`RR2xEry^Agl!Qj;K4g~V_|uX42ePxbmd0?hef zdk4$!fRRUl!~analfsRxHh&ox^_%hu8C}`#fW`u>#;vhBz>SXPk9lri(1|qS3Usb_ z8u6r1zm6mR<*<5+J(y*AkiQ~*Cc<}pEf5hvcfR=|2f`1~0A&^tkC4t~)7L~rB;3cQ zB8*5cQ1ii0T+41FA*xb%Z+SeuTH!k(E&d5<%ScHn~h zSjaLGzmpm}Q$@1(m{*PyG%Ju&8YK-jO0`+vkMa~W$yws#Qiu~dB$3CQ5N$7ylGVxS z;e9V-wO=>qVs8;_^Z@bt_8E+hUx<2=L8k^iio=DM!dg9*#RiSn(c*H#3w3 zMHsj&$Vyc6wGw<)u{hs`()u9OVeARp|ICg(}4OQXd83(gcI4Qr}2btUOdy zrCe>UDWsRCEVZ=Y%PWC=We-W3U_dQbanwPr>EazX*DDXj3h@(BNQEb{?3GLY%vyjY z;z;ga0^C`Gs@9}E&d@DPFK4vivTis7sr66Go^-=$E$##8hS$$)$$v(|@gotbZurwP zB%maa@c-c0H(f?@NT~`~Re!0-0H@L6bnqbw2oMOC384QhSZLY zv8x?z$PmQWr3+(YoWqQuh+3Ee6b?vbtFbX0ywPVINikRtR4hkG6b4i><AU+Ey5&qy`0Y=yIZlIpTN}Sx<)vB)ozrjeyO*-DgD0-Rk*5U zL#RS;T=i|lTBUl$7cG-b`QuRGnh!?DEuM;>{#;GVmBN9KyoNfAqeX1TW?ktz!1AC7 zliv(w>gS=tw?#?XK_Pw?73xV>g!eC9P}tH(gDU?pRQUcV@qAmr)Udh?{JARyw_<83 z68N*BLHsmS_=zZqIKa%d(3TInBZ_&i5v9)-qWsfP;j>X9n(INu;7&mVO2!a^*pIJ| zT5DA>-MycdW(a3%Jq%lmHVb0gLRA%M+cF!QnHqN~(SxH^^sIY%lvM(i|695+dexB| z6J=p3ClB6%VU}lq5h?mC*8)T^c}&5QsYtBCPqArKT|)Ie-NGGr^N&r(IZp@`iIO`V ztOr;5**2bu{`o6|N^{cM7z~7Ml;KQOuVt&Vqh(Z6qAC*9qMol8voXLhk7SuC&ofz1 zO8Bg_?X|}#F38o1Q+o2SE%y7+0Dd#5CtK|IwCWhZ7JK7|^_uxb3CE8_q_)_*7m*1S zMvgeQRF_BDy5UqeN=c7*Yb3nqmsggMXKU_=79N5=dM#<3UbU1Y2rG}%=w=`o5aiVFcFsfP9tJLnl zcQhKm!7|n|&tV=}9e`NPo}7BrJ?YL&JxV)&IG9mzwhBg; zHO#14Q|i7rYY@G-GWdmSWJ=Q__g=BvSRmo25Y6RjzVKB!U@;R9VB4^C0^K}2Za z*Ccx(sXFHjN>5we&z?}}eN_6PS}s+|S`YCyiuqa3WXCNJO+zUfD|`IS-W>kwNyOUe zY0MzMV(4kzd{R%n0oKhUPv}{ALc;MQ5$U?Quq;nbl^3oh%fI%=;rsxx!#uOw##Wpd z2%NQV84v(8-bc*ivC#yB>a752eB_DCR=_)^K4Ws+r?YEdN_ig6~%)uJ7we<(@Vv#uX3 z4ypCeqGYzEi7BHRpW5(J;$}&$3zs|Xar8=o1=Mk57Pq9l3=9o!;L2w@4V+Wmfb)6Q zjZyN$SE}_`EB-OPBKQEuu`R34;AKT$`<9N*^#)&|AOVs0Kz)O2U;Am$S>T2zdiqbd zhmm(_ft*q|c19Bn9;acTP^WzxU78V6kD~#AoJ!E@o5q6rm{1kcS>V+9pakw8s~|YNj1=CJEYOVq zx(<}G1NLU`-t2`2!4%({y$GXEL(TM{XW#VM>$Nh!6_$nHz_aa%w()s4v`I36l{xca zZJKOJIDRC;TbYAIfo>n=Q(jSr2a6@vTnw0dy@pS;&OpiI43~3Q$!bHn6i(h&sn{F# zhGYrgQYNkVSrL38osn3?1Jyk{EbW@=%{LJPuVnnaK_%mV2_qS&eo9Nm58pJ&Fpl-+ z{hoUlID5odBd$G|if0eV2AuB8x){-N-@@Ez=;gk}(@WV*Bdy|n%>D9AcD7JPGK>9_R={R+QgqT^S7 z1+JG1gPKwui!=p9yE$$k+cb-UWyZ)x zY!hF1PPmJ(=*lG=HfPFo`I($7B3-gffb3+q>PQ(!1-ojvD~6*41X<%eoKLa2B_443 zFr>K1h*M|I=A3fZ%jJwygBz0~un=ibBi}An@uI_YO8ADpBuQk-me=6VK{6QBmxGev z(`$AjIcch~MT40^DjoQJ%E*6*T`UsNoxx?zMv2Rb0Wo?xRh-x?S5d%B@rn`7*Bk*I zDe-Ut8FxD2yF(|9H1W_*O`r}<;C@9zs zq}{!yW$!90?i-qww)s=%v2;{jy)4O!k|+d;lx^Cn0n=4nh(_Wj&jKV?5L(=vdN!Ra z&iOJ` zpWK#CHFp8oaxxh*S7PVSFT~Aq@?2C{TXGZ~Obpb6FQ}6};lvGERs!0bj(~|4Ns)l- zn;L6Ld3Uz~HFKeB(*QiWi~p6*;=ghZhUW8~1Qk*VbFYR2IOG{ZdH_)m6%T0l21~9E zixS1Tl?_Mcwk&U&qPzkV)^u?%r`BK=BEY|~TC1ee*{iW!yWP#FY__M8aAMmhm=C zYLwc|m0>ZX06iI<@a^%1AdHUb>$Ugz#D%wObn*9~_`&K=JpsTAJ&+jC z<_;y47F6>r0B-xampk1Y>s_#SX+JoFBnK%oRn-u1bm~1m$+r+}`6-B@0S3K-Z}PZ< zt7oE_Q5JepK@3CH`$JA}i=5D)!q9Qpr)bQ=;r}I|m-ezs!Z+>}K5|C*&7_rY$2xH; zi@hGzy{0{H8a@>&8xD^{6&Zu%re0Z3R8Jh+VQ5Njiu;cUPiK{lJwNUKsopg&EhXAz zmt<`|UQQ^G9WP$zbpN6WF?Q z7I})$!S=oe!-@4C|KK$~cK1;wtA-M^lLe}`C_ zqKhx+g!9jt#pRi>BVh@3cvz6PW)C_ypbZz~4k-{8DUpEdn;Og7I}?p?S|mjUF$`7n zG%?6Yh%+kMn`qId`OxcJv1p10UKom=2|IBuYQn<`LkX@}_oYG1s?qP80a$ilRr1E9 z30ze!@87r2AA-bHN@}#MV1J>iJduXaLu*0+v2b7;f)Nz@nVq-1vW#O zXDOrzxF35*MJMt-G3^_b-cpm;p45A--pGjQMA0f@34s&bc{N*wCz*mtu_B257JSsw zj)@Ux=&~6_1``H69g#hKy)-F%al}>4I8%Vv)G|ipU$}KOwYi!?!W704u6uU-lkO78 z^K%CX@{ybS=4A#^D zei2!O6K>SyOBYgGn#nxgUT$LW%?&qN=wTeB>(EFT+)X+-9GMznzKu+eU{%S9q+8f9 z__O2uGv9LMy?_=j!t7uAVD}VZRx$HL;l`%_@j)horJM$PA5&dn(dkqmpbf9UA?Klm zek9=frp8*(9|+b77K-g!1oY~avcq1rqK+c{kB!3i|AHR=EJ|6s+oRVk5`8~(r7Yb1 zIrO5}aTsR)w4e9Z7`L2u8M(GvoeHnlEtWHhK|r5g-5E~9JQtycY2Kj>7>g|PtUD#P)uYyc zyF3rg3gCSX*jJ*Y}tVTz4!ZO0G6Hv+lBo{u(F8z0=G<= zj`)tqc3+&78MND3Z5^(G8*MD_HhlQ7l~3}PX(a;qq{p2R{2wDlP*}&hhvorkF1H$f z-S>DF7XtTqpM~cTwm2+&;X%N4Xxc>4rUZ^V4oRZDgTqHTZjF3nyf2|$f!xrz*Wn+; zmV-7hsMOg=7b$kIL^qt*akfJ)J;d6e$bcx6<0ASeQ=_0CI+mhplAJb@E}e-^aCAe^ z)A?!|zPA`4*DCBll-nutDH7BiVi2)u)JIq-BFP@gltWbuM6T>EzVd1=J0wB@BhdpzQ*_lu4Mu1zmSy#XPbg zqc^(p&IPCt7+tQ`;M-l!WyxO1pl+46Y+&aAB8O#zYeihMC^wJ>pH_QzjBfCy3Sl}F zi4+j9l=bnaE}CB-_%%^!a^jo(xNCLfMIBQnv}qR zI>MqP6qqphFW3{)9aK+QltctD3{B63)tVML;bDcLLeWcHAww$~64l`!|9wsS(x|#B_T6SJOY$oNT*+MdR)^96pDu4~z$g+G zZ)&RYO<8{mD5ql=%TX`Di9nfATT1Mx-)q-kUsOv8y1<#L%4896-{5(lc_)0$ajCG_ znu18&YU8`12p7i9Fz;I?4vk|1L=7O@%R26Q?a^@=?n2HgN=Wb$yN6&c)##QgnP(!w z9}F%C3L9W+WF$JzFJ1h=!s+GY(#4C*$@#NONGmbC#L)<~Vdu=*XJ1%aNS>Xyo=%?O zq~~Yo?O|trZh21rJ`byopH2>=Y|$Em8*oXlmZ+raxkPpghH)+C?@Zfg0nuTBnICAg0^& zI!>IXwgLvRT7bZ(&LCvx<1{=501aRgAiXH?^{cZ<5krQl6JxQb#ZK0n-l$KII*#^T8E#e8ki=T`5jQIap ztVCSFU+`rne(2CqcTv}w&f>Bx{H+}FgE*0tbBJ>B0VZ6=wODL%yhtfW4pF2|o_3*F z2iCJlpGhDIr&m-TdGtUjG|>YdQ~PVCJ^etf53O82IsVFH*RJBN*wz@{n79_li`!#k z*QguGxD_zxDDL)NBXxwzfVuT5QZ!>UAnl2wpdIO=Y3g?VsUTM zA=v|7I!aq+v2$11Wq32qII4U)D)9v6-&qy$&N6)JqHalf;Mgv1#W0~x~H(;qP3z*`whGtMp;eI(zdq6&B-pN4BB<+b_ z5$p$~7tLOUSYpJIBok;U5=l%20-7kl81*EkvWt$=N-9rCj<9%ICWK#L72aBIr5U|v ziE<5eica0LM13Z$u{KX~nrX6S_bgFuFz;ESdzNTy&l26UME5LFWnk}FqRLd>vqaU_ zth*(uU9lWpcfE@rwr%ph8u@X#hN3#>^2RB3OkltIS2_V$MZ)nrgZoY1N8VvCKE{{c zd@l~+;RB!WctawE|*>s#kC}AAZA0UXLl-X?_e33Hl*LJ^uSDvBVv)DYeo-KjD z2*t$vhv3nHCm^*^!IwKlH%Y+0hOfx@M%W}9)gr##T`#ATbm5GCMRx~)S9N>0d1MW# zgyGiCT0fqCyJj*3d>BUizpxfl+mMn zK|i<)7t!GxOBNBJS=>Fhzv<*sMg8?a^FDag!7D_ZG-|5Rf#!iscD1n%c!xJ}mH}R& z1>Bh%yNiXtyno-mk2DYSjqW5$5R@rP&u;TjqrNur!scAM|={o(^r_zleV=3#bX1-$)`_Xpq0&3kD?YT!@~yeT5X z%5L-SBAlNMa?*3|#^(NPX}9?{gp(qkHkrv*=qfu>8ZnCx;@UpE? z8zf8j!i`?8Ws~-sEFS545*ZTmLN@P7A`t;ytOp?y%iGmV$UWA)3&c?FXAY(X(7REKIWD-F!s8s2;xe-~eqMM_^KIQXtKazWyH zX0|Hk6Gn^_52|riuQe(_<}j2F7m$+k{G$+(G7@>NVW`!_KSHViiCMhu?#)tKEX>7i z^p(qI>wtEDx>!zKOJ+Ax`C>Aa$CdgWM^M;Df97$`p)f~(K6)?yd?)>RFMWNA{=D)4{*>s?a}VOr0{wZ;rBC$dz;XQ9 zPk+9441fNL{`~v}{P{Wh^9+`u!Yuunc^-cz>Cey-{=Ad^{4&yp6#gmw`7x-a!r!Go zKR|zW@u$%|M2x9#S1>3$yRm|^vTvWDeUa&2WSSS5-bJQ$k?CAy8W)+qMW$_$=~`r( z7MY$!rUidN!XneK$oMZZ?u(4~BICTs_%1T8i;U+Yq#iNlsCx+q| zX=HdPew{{!hhmFHhKJ(QG%`FCpP`ZAq4+%-86Ju!Xk>UO4${c*P-N&64@HwkhKJ%= z8W|pnI*kku#W@-o9*T1`GCUOTo1|Z90PdhM;z9Yn=kSRK<+C(KJSgX>2@lE%8Y3PQ zDlj1ig={Pylmf>RH1D=G1`F}sX(Sb34FSOxPe`k)c4D^JPe_OHQ|OJ2)pAiQA%6IF z-$UIRr)a_e;AV$Zul((Z2R>HjYMo4h`~9%wFZ-Gf)|@no5kuqbHt)p!_@cN50fXcA z)AITEL-q>)8iRc9mfDARmq^d2aO)z8Be)utJ<+jVf;^A!)Q?$%;6NLzqFg}8NNe7^ PhLS~=1|hyRRulg}%WQ^$ literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/.doctrees/chapter2.doctree b/doc/LectureNotes/_build/.doctrees/chapter2.doctree new file mode 100644 index 0000000000000000000000000000000000000000..f9dea0135100cbe6fc076a07e49e4ac43b61d3b7 GIT binary patch literal 169250 zcmeFaeUNKcavw%6KOaagx!he!EAFnakB`flhk1kV0l-~S1P1ek`5w$*FvCZVE`W<~ zTmTmrzyPnFC2cH~B*T>|X{@pnM{%iCN>mlau@a{oTXHsvT`JkJC3)q@ilueMuGm%a zANf$ZR4F-8etquOxxgI&GXR$>6{p^t0q#8?-F^D>>F(2~&ksNIm!5p$$tUP%@N%nB zEJ>HkLM|uOb7Jdi@O-YGZP!Fex%%m=#UHtPe6=vRBMaSDy)9?OtHBc}ku8?0Ia!ph zKDZiunu;r>YHL}m|6EJVDy6z)R4+bLe75-92UiQlJA-GGl2R49SA$oN#FkKNR7+C9 zxhpEgdaiXvrBV3lWAT!pur2eWR}3~Z{@KB^4FS+7SH;h?Gs3_jD2iOlv=v}O&_7i| zlYsQOY*lEr@DQJ$k%XH1>G@o{VSdR7*@`UYsZ4wDoFWwH)m8E3!P5XxyxM*Bt--T8 zB{tnBEm3Z{YsIoCwM5B%EDE5ERB-RIzv7N4$P#@AiQGG&N?R89>Hy1~EeZ`qlzq#k zMo-FI6<;sD*vvYljtYjPnB|F-C55&wN^e@C@M;$^|ssH?RJ;jr8tTDOhYJ$i~7ji0gXhE91Ni|N%5J%E5l%_#b2{|VT_*5 zJ5VD~e6sjd@%~XQn3xF>Z;Nw818cTeQ7u6cY7u30CPL@~n2ZI;4Bt@?0+Um4oljF6ahXy^1 z2Y_NM(FlT<0m1j^FmHqP9_XE=f8~ z6RpS1Vgh?Z|6lYy>(p=M8ArR^^Kmb6FdPYH0PRO$o zNg;s%1jA!G;G0v()Z2NUd*X&FFV)kvQxYyIV3KjfvDf9)Tk`Nhk<1fGq&57IlDTmz-VEnH40*5ZcaMDbHq&V#S_X*70}d1Vvb)R*crs zjM2JMSMip_uGYp9qZXO~MF2nGVUR*_LLh=3&Tg@kEjp+VfY`1wlA`&R2F>euZ|inAtj$FfaEQ#Q!P{XI zpNB}|keN?=x5H(9K9Z5cW?n7c4xiOIh-eO(xioh>Ok%Sy6dWS+Xzg}*tj=Mz;t-ij zbGO4}Vd^#0T=S-{GYpe>t*)&ru3s?JMon=wRL8;UakoxdhHlRBx??>Lg#=4KR4!;| znt=p!MzxTje+bncY_lcZWSC1^21Sa07l{4}znouhROpD_8%Ffb;AK&QWh|2_dr^g^ z%NakWiF=3MT$B{CHuBn3vj^XM49)2N{Y<^Cw4nVooGP*I-#1jBr>O+TgX(wS@jE;o zFdCml19hpu^)`5AxLuvBqc02^*;RVI_%Ha*#uM!iU{t)R6O92$szmWwqX(ulS)iZH zZbpN@t+LyA5Pp*f;eS$V6n{&%ZPYR12Z1b&`}bGndaLyyB~(k$zM;hPgFq$tL15pD zub0Vx!jq=>ZM~PlE5q#>#2*YH`os37G=>L{Re*-?7Au!(Q2KMAp(3>s( zztfI_Q9~Pr@qPRQe%!XTaWe(W&u^Iq9hnqli{gJv>|ip&w4g`Y1Zp{3%$|Yeg6Wfh zJEtO*@?^W_*ppbbo~@Amp$VsD=c(xIma0|QL#=GP)gti(FAS{Mc2$raS;iY3gEx$Z zhG&>*v-CbTXgNmx7=gg-cre7cHpY`(nfo`OwmEffQp*Om@tjG%nnhixjc{iAZ#`x4 zmgqJ#92;-|kX@Xu*BWi|i;yb<{txkzyeFj&+zc5pD_}t+w(P;WR$V2$;f#e8Os+ds z!00N^I%{Oc!&BjeYXUtnd)mP{7up$EZ!Pjr&>S#qYjQ0tJ2&fHnl?+$TpRCbz0Sf@ z;1F^h0meCFe-&n$Bn=csQEJA52x>LkC0QglF?9qV#mK5(5aEVoIMZAuHv#+$#Zm@t znU1*Z#3WIQCD2KAg_6?*IH*%lfgYhjU#r92gT8CzKT!3D(Y}_Zn5N6l803UL?V0>2 z)f#nK5tyH(UTt9v7^V$bfTt}BdSf`N_`p0zfTt!%Jot)05WAa|I88b!NscWZoP3heli$-TFp!JO=!PkRtmYg33(n=%6&)=a89GgR|fix%G;};W0xa zv%nr7SzB}+*N}LV*E|l5`Ls6+Mv+@Dz#JMgG%^e9YjaTnI6UUm;4C;rXI4l!eCE*D zEV!(h?(-Q`BMzTAH8=}SajRN%J4KB{X+G`Eg3-%Lud|- z&4SApjKzgCZ;Lu@TI6F2+4Mcx0DmxSfZrLskZD(|qH>)Z_IjUReZMU`TP4`g%&s#G zbp<8~nOm?+?yI&gd`e`k%P_KG<=&_11(fm6bPvIlV79RUGoFl+-8v!^oU4Ldt?QPi z;m^H0xnpjarzs6u{x8E8!kxkMVy#gWTBR265dNGUO1fXt{-rsb_V3aA|2EY7SpQ`C zUs37B-FI~tr4BpWa1Py;ye62pZo-C+lX=oW<99V8Op5v;-?UNbcZ^D7+|$Kl7|77N z;qEcmRNE-g+ycaWBVHnk%4rcFBMe22(I79AT9K@4ld0k2U``>O5x#RB8lzTrAl?WA z@I3vQD(Fk!pLtK`qi3|L(+q%7X{rJEi74}j!HLBYt;OPd$fs4U*O@DsW@FD`s%m6z< z4>KHju7w>E2RbmDjR?9aV)$M=5)qt5F=_}BOX@ydE!6e3!F8`qBCo4Nj^ocHMyKe$ zxK7$At5E(Hh@kIK>3fw!i{t{gGj^bIzPhtw3JmZ*bAqyMQ05}#Cx){qAlQ$bw_2nv z`R!B&ep?qY^KaN7(O~{u4#;$QJv~xoz{fSn!otcLG9Q9u{7z7epC7@0} zyH^GU%601T^>XdTFmLHFmkpT5BTB28QK?{58nc>xHrZ=oQyf|sxz?>%*}gN;`Y{!) zKYwGiwso{9s&(3}OQVvBmMJTA*`S9YLl)=)#vq3|?qll}iG~KV92pRfq2cBzW9S=A z`ZAN#;w!OlB;^DdyPixPU{RTPrUmHg>@>dy&C&Wrb+rDo8;{nWK3YF-&|)$Z8kMFZ z;JFPGvb`W}YYnl@*;2lgWs7&Mjr|zx;pwJ<;f#jb&*+&Yy&tkt|D78U*{Ea^c`R$|Q^MeGT7`zU_0vwQNl!R_vg^Gk6LI9(YUqxvytrr~6%pE`g2j#@wbbsq()Y8rSOu;ow>D z3zpTZ!P7;hR=pbBxvW*+ZZ!ltt@S+31X1pYbX2AMhtw#~GaUWGx19qV^={vW1uFH} z&v;JeAYYryB;?STQ-ibM^!hlFU~tM{Y%!bu#|mwxydLSxgPp`4G`AFTfWa+`I7kYp z!ZDD04{h%=R@Xa1Q}GykCKi6%ug#lyuzy1AW@wsE#pU zY-+oQK)QZv)0}-kr5AV4^>NV=VqUX520b}aa1U(=P+mrZx+OSxnMWPNprbeP+=@bM zekeRm1J6OPezyVKP9EsQ(2X;cdDVu84B&0w+VnMfo*3i7&67I=2eH;ghGGnP2K#l# z!;FZNPLB{qSdi{Nb~+x`GWE+xtM$&~1?vc0cpoo0h93DIFFL>Ke8>5)D(01kkDPjA zvMlr-KdNT4kBu3Ia%$71{A0&gouvye>Xj~hn1c_U9{w#p)7ys#%wi>=20g^gdnB+{ zoeyy|tEenH)MqN##+vnK1|Nel*u>0(tH=3P2G1g<39;H&KY>r}T1J$w_R7zoQu$e` ziOHIyd9k>AcTYVITIQI;h-OXqRJGSBLoux5tgaeWWfi0MoCr>LW0$B0$6w+zzQOJy211wy;D9GCcnXNMnYVx}8FY zpka^UY_%Q*XuwKGED>&KbgI$=j&4b$^@vQdDh2Lpk#G5u40WD7QnfnU#WD5iSD{Z2 z4FM;G?A}67bZC4zFSS*ezQoVlk}gVzo?2P%hlU!QYAx`!HTKfMVQaxl7M9^~8G2!Z zhMj2D+tx*^71bAchSV{&KC~kGGxJ0A{`CCMQQ;=FZcHLw8=a9zzcfE0-6bN8pX(fP zrYUVh6f3$`m1w-|&{^h2b*voq&G`}Q8Ky>Y^w1tmM0ru41`~y!j_?g?Ske8~JkcdV zY(bYjut9gcWYF0jrB>%4);Wmg(%h}_vGTQ5tm)|&^pU3($ON3}5q~(pp<;WDQ?G3X z#|&;(&xTNwHDF^EvQu&fzkY<%4cIr;Ohir@NNb7!c9NifTr!^@g;Q1Zp6$LR>4=JleqVQ&pTHb#;$ z)aQqqoJvcG=vsmf5YrOApDmu()1Jd@V$q4n-Z2JJ7k}InD~b#AL-85LG#nz@0~>Jd z#G6hoUGYJgmxmaTYkTCCO)pk)@>yMSaNMht$M&6-Q1ST@>MIlH42S3xg>6tDPhf8S zc!8UpT(6eLHtkzU`=p_CCl=VYELek zZ4dCp2CT^ogCThA7KZd2n)S8oLR3na1G znK5W91-4BDgAe<22ye=_t8}bLqL?QmrXL1griAQOENu_%#9)5~J=F90MI^*gR|Ry2 zlTIxJSP9O2TVg4eocHnzyqLaxtT`@GMpdbK<5}Ix#-hZnahxIec)UZ@!SeGsk$EkP zoMiwG(TNn3Wvn~jAUpNrB^JjM-wn`!DF;I=ZzpDK%6w_;sB>seC};zLRSUx+QD1R0i86rtZgOV-Z|$2(&y z2KH=;`o|cRxp-D%e|3y!b@0ch5S5h1>{8F!hLysQ>g*sb$yYTY8H?b6U4U@QDiT;S z$0FGYHn9gd?5qZh(@|)w8)y~nIE%pDfgNKZGIC_*L^yC^{;&hEEId~| zeC$MMV5x?6&Uw$W@bn%(uy11VL*G3d{&|3=(^WdG@p9LYUIv<=Idjo1BdO{ViS>GoO)jAbhtqcD`{SvA8B45 zp5sPOgOOJ@B$&LG7>sw_TJm+vS^~F>Cc|hql3QfbkNI!vcF(f0l+4zMfiqi4+@S7U zR+7yBK5ivBf)FJm#N1Azm0={vpPW|~Dc^&1r?7oJUjx!C(7qEnoD2bdfBLpL{=Z2r z@WKxaaiLX!S=>!MdoL6oTS>n;)Z{dCJ|TbR$tN^I>mV`V_IX<}nLQ4!DKH7Oa)tyj z<&Q?b&S59O`*Xu!;?&|1zix8A{lhHQKVIK9_HLIDvA}`@h5`Zp6k}lJZD~$oCGm~0!SCjpR!8TOi+2HFF#{hu3RUOQ#Srp(L3T^WcEMqb(UGMUkT>cR$N zi$#+D9+)S?ysn0+qy*Vyb&g}52y=NKQ^4_nd&uU*2Os*J57h_*dlshI<6Ll}GDLUpGZ!q;U733Twl$yMyjhKs!-ow zbztgE)%7)PxSVK*2S$?<`A4c-d7N+<#seVPq2b`<3$WK9XH>hEP}f6ikz+jIi`Nm| zrY)!I+Gf?p+ey+- zrUgR zkhUdC?uYUk5cix5l#q|*CzSFm869h!xx;Y&hEiYZyWlW!=y5U?Uo>~apVJ7#X)4Ae zXc50Kq^nr1kL@{9Q#aS=+NPD^{`wFyQ`tSg{7VLN>BL}mkHZ{&j&}UhiR8PL@%C%; zMB)t?Z<836!)~hLHfY;vxH>Vp-T4=|8Nu~xd2G|Vl_xFAhTT-s8EtYwEBAbRo`nB2 z+tD%RGl%W)n~4ZNugd}*G4p{Hf&Xis2!w7ccnp{Zw{F6QB$KThZXVm(bNN&F-kE^h z-${~#$a3mjwZuET)Ihr9Dm%b9!TYQ*6QK4PM23l z{e3zg*}XDGkFDUJ9cp?if%@N?0#yf%3DmFBJ_@<&CJ@j7VK8kC8zM~R7YbR?dF;t!Fff}#OPE9R@Pj9Q4QW)ydBQhX z)+;W!k~zZFw8e0AYuX`qKAMwv*e7jE6oMn>RMinH`rO(x=uv)`2@t<^j1@Q2v0by#Eoj%c1fn^X83^b*4_|0QUzPh_-ZLlpLWJ)o{ECXEA=+ z4bV&|@q>59)@%Mr8jP`I)jDkS1~}2}$9WPyDmJ%)A8r)ObkVgQFdTVZ4ffNi%{`nF zCp*3!gO4$JZix*J^BPmgz8m|eDuPQG0;4*6gdGaa#iMVM81Jc}iBR4UeG7Z?jURJ> zsaht(!vN+(6fg%&9h*^saSiUQF4uuwV^&}Qr_r?vjybcwO4}iT1^*1vSt&IIGNag7 z0T}N78aXg31dciRJo5G$`jB8Jq^?1S$^9X6TmuhiGSVrNOB%tJ$WT#fcRl&+`okG_ z+E3;dOHGkVf3lWp;#eD@Pu9|5dVI2$GK#1Y?31-rUkWKPh9TALC8PaQy#cQJCu=Dg zCA8g0B(pAMpRA=Oo2jg0aGj><+)#-!7SQ>vrG}@0`HK3mtLNe?`j>YHEF%y$9G<;Jj+tx|l~^$`XP}-$wcq9!7@VW5mDMI0r7u2!R6hw&fqTnzHi9 zG0`Ib;Nnra{Mbp#z%aXLU8E(W}(ql^edRm`-;YepkX~l^j1kMBJ%7a{u?1 zAtm&|VIH~Xw8_Q|t?_8$K<>_UhU!Bw=T@!6`-OQD&pAn^a8#XK)`r}6o9+7WahvTH z=Q2O5^_>77j!9yU&KD2&VV%EEmi3{+Co6e>ZJy+PeT=v@`yGreW`oRlqUxszttfZw zh4n`*+;uvCT-{p-_xI*GxSzLOx42=wrogQS6vOVj#j0qK;8wM5pq;$A!O3cKbHj{* z8~iI=(~9IMIGjikVMs{K63ftbyQ8LK;kdkd$RC&7RO>zGaCYV;H9Iqj^Cf6xw)%+~)E1i^ zS&pgZ(Hh3|kB4T0ok+1M^X61k=fIt^f(=Ag@nskOp`FLEdu8xEvs7NfMjD@AZz_G` z-<$e|P6(!Ne3k`Efde`I;SjccZG+x;g6lFIrIH^7ZyNOZXqvadce2pn2yQDh7S&X} z*Q7AUbZpT*ICPR)UvzFz!yri6OU&|m*?kk9VcY&sBn$=Bgx?aQGx%FvfDWvuQUS3r zybZ+!6k>76oRn*im>7BtTNWk91>BHVzADr*Il(Y{$r5cev``$YX*m+ zm}v(_akcn8oj2ZfdSGi4K96J1#u@qCCpQ!P!e{$+#dbkeb<>5h8(6=Ti~4ENAx zIIN_8fnyb&CYEPb;A`^)-g|S_G@u#WqOuJ!Ca+N(&F=p9hF~#S zg=6pkkW8l%%T$9KZD5B%>B8csX0K5X^ccpMvF_YMRyE2vUWe}%_7bJyg!W8}k2R)H zHM)%YEi^Eq8<}nOMY(R;SW7nnj1!JGtB2|L^F!0mDnFmoprV{$T ze{GD74g?eW;oFjzBg5_6m`D;k>uhSn?VIx?)Ta$Xsk~#1$zb@64N{Yr25x+8=gxw5 z#G<6$^vTkIa0`Tf3mC4~(a1A>1p|}vKPQ5{UjDtC(2Xq;0h7B-=q-yx`YrWde=u*l|^G!8J7Q+Fu`%c|9z)Vn}Yf>wDs65&o;A+T|oO?)(=>Y0PFHI zb@%GpW0*^ykWAa{x3f<*MZh4L!NhDy`#0d=xh!e1m!2BCxvXECVa`7?ekkp@Bo>gK z4T+(do38{#u2rHPNKwcZu_IY)U^f;yc+{f`bVUoLoko_p9PL?N*T%BbVLAE0ErDHq zZ&y8$OUom!mQyc~qf=nr5^_ijRd`F)`{yin9qpXLms&uoX{^Shi>vKgsybVd2GuyF ztFFO@9jJM@aqqXf?@BM=10^RklGo9GEUtsrT&H|~)K>igrRhe?Hknc%AX6&9#ffBd z{a!r=wlfX^x1PN2UvTIJ4IVp=Dp!Y3A*gtQex-^as8I=;uJTCzj`FL z1iBOoa$UWx)^JeRy)p)(yiGjzdU*?f#x2yizp99v!xL$9TCKAxoLTvYOf^>Tb-=xn zmfb%+M=En|WpLpsvm0qWB$JP=R}M3viR~k@g%WpaA(N>>H@-UAz5wpEYPINAEIo-H z%h#%{a8$(s7cE_LOqQykr~Bp-xQ<=aXiZyyE$mk#>K}QuRlz^r$&K0-Xl!ITE>$vW% z)Iwr@kzGsiK)r-Hr={0lK%(QtCEQz5lCU|$!b&NwFke!bqKkf^>*8a(-fkI*nQ^ys z6nSpwZslyTUZS8xx*-N~E5njLYyFy2d~Gf`aGW%!1~FhiGR1_qDYH5UdUE*8rMVkp z!%cOMp1XPKu{9^i&Fk7gI(ed8OqeJGgqhH&Xfpx}7xwVT%3>M8$#%@wvsm4rJ5Vt& zK3k*^$pV)vLTy-bCe^~lR0)~8H2313&r2E zKH!wMzmMEBxsrOr36r-@2vCG}HP~*q=)8ZQu?%~4|31aUtHBLxvV8IA7~F)^^-$Z` z1ueQ%$E7KyOUQJC$6BJsbl_aGIbH5DUY1K>-h$J}zfWw5PDVqF^p(e-f9u)roy0>VK zu?`P2X`kfxiSfk-UaS4BUL=Lc6mIZz*lph#NHC|~AKV0T?TYGaf&a#L0n-@h4Za(o zu^GO}^Pl65U)6?Zk~#OAR+jqPMzt|gB5h1%RYnYn0a{BmIZ%RkRoAC*bDmIuW3x_| z8Phd-v~n3O(F`}0te1`FY7g8*9eD*x{mBq7S{eOwRw<=|r*pdfOM@2}t3Q9|7Otm) zIZt))*gW+UtKXC^20VjXRktD0WM<+J`1b}^jF(TuOk6dSna20_53D>$E`?#!A^q2} zVHIU}s%Uq!WJI{_;j*sPVH=S9fPElWepS4L6+n+QQ^Q~;wu~q^RnGhZ%tRoc0mp_l zEt(2;fRGa!P?+dmXf(wdE@g3RbPnzsI?4o_ggoi0o<`hjo_u1qrs{bRvoh=#I_ks< zk_(;r7BqE^43umgH;##yBUXt58cDRRdVa-adLh8WBxSeDv}<(BH$Y;W70@HG1Op2L2o@{N2<9+Is?JZ= z2*?4z1rUbagk#L+^p`&wI}Fvhu`}h{C&Xg&WGt~D&K?YjWINxMO_|DfaE_Q`I{1!u zdt>Twrt`h2=C!#RO;_FqqVWp?C0;9;>!;m(0DTBWKR2OV%ksh>?PE|9(lz zi4Dkn+TYF8>k6c|(17tmYX5taYX5nz!b-UQ+;l{@@VC;}rrjSnB3L>ZD+b&@*fVICQlK;}Ltw zGGIDFLr+aahF+j;Wo`)YH=&k)lo-r^#*elHRKYM2l5D^+%+vlF<(%RN@&$crKsnb zJJz-iuFb)fLp(59 zbjqga)R{x|^>BAE>yk$=pdZAZEP1%r?Ich75^6imzJ$I!GJ$PB)4O%KLPm*$d%fhO zG!N|HC)|f{(jllSgBbdpQ^wkhlp0LQ%T&eIvSW*F3PGA-izZj1t2QC$u!Li4#sQO} z)XJAgkk9~W8{4W^J1>-~xG?2RRM4A3r^eb<1S-+7m=fItO*e6>`%TRDLz7cMh%=&) zcJN@ss0VqwY8aql5HG-!Jo>2Ppyf@3-GgXrY%q~;@d2Tua+&sB?U2y6DFvO*HlFFvc%IyZ8gZ8#9+!$s~8ofS- z?>NxL7MrF2z4`Js2lBjIo(0SGX?SDul;g0QSBtY?^`c>;LIdMqvp%=!!RnoSk(V9% zN@IO!EHBI(`A0(&bnG&sJKj6yN}^9*wvr%I(e%QKL-Jv z7P0Af^0LU0%63^KZzJRTRb4??>+2dVZ_pe zJpD!7nnp>J5J=1`=uFEWF%r#REqY=m=Dg}LFltku4T+c9@Yz|Si7)9Y@P*3@@6zRkZ|Pw+2YY!beIYFpj62x48!CNKx(AI3 ztF&rutGzHTrVZnFW}*|vZZkFnKKO#e>ch8A!dHfeM4yPGFATP0lkx6Qvr|pRA2KJS z4%W45onAm!>+I%@O_~13$4}g_eVkBC9P>B%6KU7-z0L?zmbP-oJKs6(;3Vdg4s!?? zlbZa&m3o?HvsrMUexFFSSFBXKDCqLui=dA4;9Eg2+W!`=kN$8G-(o<&?O ze4$rAfluvPMwGAi%FjIY#1rLb@e8rOVh@v;iz(tMxGgXDaDH23rFdI!< zZH5+$d*;~`@iHqm$jno1G<8j4OO`_gg|5RU6R9fZv8uI|W#gepyrMBHGE1UhwJPGzW zyGhB>#{T`B$S|tQX{U}xUsy@-t$C8*m0^Lyp+EZ026gM$-Z8|LKBU|eD;ocLo@m%@ zehkm!$qOtej?9&1@KO!t&4rX%mJk`yx*9ycDpsp|_rI`R7HusSp1rTZ*-2m zwID5ijrLz!cp#vog2E7ZzFVH{Ni*rEdwr;Czd0PqG2Y_#`Wk<=*3rKp1{3 z%A-VGf9HP~%;e)N0XHHWLrWzDL8H81`Tz(Lc@}bFRa8WkV5LRpqC-U(C?Sbp3v2(M zh93qq_S7V0;qFlzF7TT8w)4#b@@r)8I^T3Is6+Wc@hoDvwQ{DHJxR(9T+P{*^X>$!D!*D1t&pJC31#iN4;}4sR zCekY7X(%z&2?Jp2sI@UkDHAxj(d+RnIlVy#y=&Allr$&t(BpAj;!~~dbg5) z`)%bqMOrQOIkQx(E#<1BMd=?F>!Hav^ zSF`TI;03kfh0<%F$g_wS5QW;+;Au&yso!%bfbWWUi9e4G&C0Xv^{@E+5z?<5_Ii|K zO4Z5Xj6{|r-e5oTH$MI>3Z)aTmCuD%WUe+x=v&G^ILuOs^75%~BPE3#4 zGz?txi80g4AE?m2k3Zvlo_Yp@pZYCh5~Nub&Yk%?M*9?sK`tJFatLY3?Br6`DKwB8 zh{ERKVkG^DY3oqSkrNMtk4!p`(zBQPp4<)HUK` z;1m_A4&*n0J*ox5u*ka}zSdzC&^b8p6%Myr;Y8K~5$B@;EA!=2WS)@gAan)tr1dr< z9_`7+thH`~9p|x_M{8h-5{(`cYI z95(Z5Zx)PR;F-^urvuAloA#~3g_m|Tbty3X$~twmlR`H*yD39RKZqun27 zZfn7Wi0Q-O`3#~jI%pm)DVFh!N~sPEnDl)Xn%5#=)D0kv?4MEWO!oR^mA(F}8?)Dc zWU|*U8GJIHD~Y|hYE!Y-H?~+Bz6K7DmY^rm!iz;1+(JWV&Sic*1_zXoVaQ06ql5|Us2_AUBnez{;_MvqKDYEGm-y1)h3^JX%z^t2b>M#e#sl|P&4K%S zHyAjs+EfGgNF6v?tU~v}iNv0oPGH;!Y6lkCynR4wGD4SVr?>&@hz?I`#|P*fk5Zzz zU>2;Kt+o*ph2sX=Ol~b2BlfYvU<}pZV+8|C?YD2tBR^~M$UnaUk8ssY9vQ!|(-B~9 z$swJ5l8{g zJx0(JWKc?QZK4ryZc~}|d`S4|GYg@9Y8p>=CTD~~=-#*f7SrSRS?=*6ujCi zj1|b-ERF?%f&hnRH9PtYw?S8s+3k@A8C0s$Do-;VwiUqUa*W#ps-ef6@;XJ4NJ^Cs zWDz~9G8Lyi>rXIg{-2>qW8HoA4|ClVjHsKjx2kJM6Bgm~sTZwg)7W)a4+Z{f=A_n< zV+Pk7s-&2QI>A9a&k1fS-o4641|4Uc)8}F z7=35cla=f?(^wO@HgBvG{Ljym@NZ0^VsMyFRon*o={OUmICX~WmVau~x|OsJ$WBcS z(uF4f$V$w=Hcw(Am;u_Nq$xvFXk8|Xf9a_waO4F)gPS*ELy@WJ?J~sbWWCRXc5tI= zHt<igP$s8i*0ehsLAGVx4#CcxwP@)*l z6vLluAWV5Qhi-ysiDXzPuwrn$ShIBCa?A*s5<&@k*d>=*l>dNNm+z$!G`>W(&Lvqn zMP#7Ph?sNO^XwwDip@V*IbC)}nLLn4FpbwXCJ$VwC=PQd(=&H&J~Kijjt^{JVz9nF z#DZ4dzV-V0Z-$zk%I^3tOuIt|i`gAd)4+0=sNZdXx8r2JCEapBnT8G!%?6V=nn6do%mu_J!U&pkcTH?=6 z(YJ}vy$WAds7KJgCv9L%G*FRc}3zzm0OW+ zu!UHQ%R-mDPgR_cB?pA?)prU$#x1Hnp2(L#C}-KR=vo!LQo=4Kvtr{Z)TCJy#=M%2L% z?{o7aeA^B=cDj8VV6MM7G}hMHbZb59XNH=cYL@-i<}A~}Vzcadot7NZ=Cg@NjGPqv z@V-7R>|R;t!Y|Gf5xcpdV>0JT?^dhZr&z zvQ9B4LI)9k6q$|tfDkB6J{*cX!(EwFPf#3c&>$tXgZIaaR5^Us9K#tzt;0#LMHr%E z<`QCJS!M~hmz?jf!FE8pILYaagd!g-6cwe>dfV+rZ)F6)i0y2(lv}RL1-JH5sJGl+ zZ)n9I2n2(^MJGgY1z-R|2d?vkK(0$~J5RL?crZao&!X*o)jGL;r^f*qPH1vD83?!1 zJ)BIWXeuY)}66qf~( zu<%PW9t(l~xU}>K5P5-dky>3`CIABWkVw!lm7c-?bCZRA#Wc7-#I1HR&8E!i99V^e zW-iU$7#mJPb8O3j#Xq%aeMrWRvb&)V+Pt#L*tMZym}=4cJLaON4-s4RY<)3$30)zy z0zc8dP0tg+Pg|!yNAck|HdG(aq1>2$&Q7Xuec8Xa>(Othr7u~8l0LpePmQ4+481_x z%G7i$t`6KclzDWhr09ddzm79r;DjPa3Cm+U0@#G8nEb?%{7Yj6E!X3Of;KSRkY9gl zYC6*yp$AW4&Ay&&-*^4uz~MBmlnp{t^YLeG)*r50+t)VzSqV77H~e4DlYC1zStvQo zZdT6*<*7-`uIQU_b>;w{OLMc}}`CCDDrP_h~+L8}8CIiE%}Y z`KgAwzI+Qa^E%ReWg>HPRGOmjoCsz~1DCc>1Bb^}!hB(#gt4=~b*9oa>B&-#BfG7X zyH9o-7n#8w;89cQR!bEM(6EXsbbLQMzt`8-kXq;okjeI=G@~jy}?GJNIQC? z+;~sAr|+2<;-1UfE0@S6F-_`TN-@a^=GW}IGg)eIT?vq9eTY2P+5V)YPilNJhJel8 z{h^9eiM4-Zl2-4SiM6Q+V*-Fog~r}2j{DK_WkX2Xz1Mru`P_=$S(HGXS9t<@*)s*;6 zb4uu7u_^K3M#SMTzG20Qvzy zEt$IeXGVbH7{=x$nBtp5BRHoy@*6|VjvWBqr}5k79MQpIbL0-@h)PHf9q!>oH0(Bx zrruNf{duDD^3bqz2#vg&h}O$`$c4r;`U*Dk+$z8Rk9i{Wz1Q(6huQUICZaehi?wkZ zePKoNQ?J~@=D3al?Ie@Vq0>tyuJvElhG3cnnX5LHWcu{g# zoV;*w9XT~Xrx_lunmIhLW>4A;AmqLTW|F?Qp|EeAf%#M!sYebzi3kxJCF*Ue%4nS zG{BWL6M#@z0v*5_p)c6)Lka;s=aRO0ThHUD5Tyjc>F^S6QD(IyByo@pmzTrzpIwWl zZBz}zQ}bsz2pT*T*oD8+GWo;^PFd( z(%9%Fu!V~aPpxs?`Nv;1WDJBuTjg>1!AON6DlNLA43>^j%=wU|H2u)WGUA)Mvw7HE z6a7w$9&fskPQx2CNqpCZE4oE*pcT} znf4p=WG`&$*r@{?6|O5W5j}3j(^Lu_L7O*L6n}r7D1Mqkm`z27Lv8qt4NB9^CF{5v z!o$>*bjk}?-CwIjb_B26vzP2<0Jg^rB@>zQ-$CB|dn`+rp~&7-wbFOIK^Ow3`oUaS0vK97 zrjm^@W$$x9=Vod#4Z4{mNxu_<#ZlTkN64d^bYEb(#2j?&O2DPOG91kXBqNiwtpQTd z%~tCTWPwx@uPz|Eon7XLzlcv{M|N2@K`q0-rjxNp)Hr<9xS~Qm5vDKOIrW|d?XpEg zma>z^4x9c|G|WX(Cj(9q59hg4??86oy5p`IhpLi_9|-CR6DO0O-3GJ7D39@fgw?As{@G3o7KSwBS<+)&Xf~7B}OZZGZP8!hpncpJdz$G z^cATUVb{=B77Z&#PTh#HQk$jc!&cTpNkYsh3Tj8dSwcqa2n^RW@;DEVgBH}FN3V{8 zNKkYs92O%fUN+!xH6t2`1W5hqtd?peg<@ChvRE^HqU*ZP(1 zaKmM+;_Z2Tj_N|qd|(x2-#gURR2I$uYFacp7|f#iv{fK*#2J2LgQOLKJBBdPG2ot9 zQE<%@g}akQ7l+m~rEO3hKc1Wswl)_j$pJj424}%3J`b7Ap){ZNX2EEESa3ls-C_aZ zaGF<(vtSjU$9l%$G@tfn!DwwRiz|oHoEn@3r|+Mz)N{~IENTOXRVY8F=?%I_HojX0 zvNBJB{6*&&9fiYgd^sD`rW3+e0=zt=66u5(d2S`hUzjIBp4Zhd4tevL4Wbql%7?Va zgnRY~{o$ydpPeV1Z%wHqa%fCl#Rkjih7+aY^Uz5-py$)xEEvVK?T;B0WDcKsv^EPi z&l|eBF^LA|GnZ zC8S!)U|*+-C?^MZwWNM${(UN|BBCaVnE|H{LA*o^0&7;XlWrjJ5-(epMJq zN-dZG_LiJ)$qyc&S85e^zAMFdyo(Nk_7ZK2)rGf`HF8qEr9SLw4-d@DecBu0vea65 zAGdo2am%O2v%Iq81e*S34<-{eSCC7&g=4ub0x9}V*1&=u-}{MVpOx>%)&E4Ye-T7S@x8oxB%u74VUx3PA{yzMolht?d7gopb7`CzQ zpS8Nm%U?#h>n`a0xj!=QCj4!!9bC})25MLhkn)1@sO-YW53UxB`z6a>JTR`%Ec>-@ z>;lblkiPw>aebzDITQ*;{OtD3n5_ah{q32wl=?g&r_&o)j;sV$bUCj+kLX|7HJUz8 zkX@sx-*-6#{4?zGudBf`qI&=5r=yA@m*C$K*JQbl8#bS1q3^gE)b-X8c5`rbYE^s- zr+X+}3gwiA(n4{C*v>QJF%k$iS6b(^y)v=AJXE?{?-0qv`OUArar0lV_`f$u>m zRgsdoMof%!HP!Z|rdrcG6=a^SP%ybl%i8A} zDd605gVOv%qx-M@fk z4c`|T3EZp6qfK1KF=&7)>*UiK8e$68zR47})(8i}*pD^GRr-gYHYVGjbJrV+8$k!y z=FbU!H(WYy?ULr@up8OHMVR{83SGDAHa#&YLPpGTqsM?Zr=vEyYz|$jR@f3%E? zww557@@3LxNXU_tZP(gWY_-$vbeX!M)N4pE4*sA!cF6`S6=2E1xdCCpHssM1Gml^? z)eERj78^d3jYMWP;1qMJUApAFf4_pOmI^I49-K(j-GDufveOrj}htD9Q-{?jz4*Ad1R{ zy7Xo=eTUjJ2>T9$34~QCti3eI{I(%l-gTlmVp&>&Sc962qmUIA7F7$J3Uku{6;|kp z-aYa9HL{lkl3JPJr4y|-^xPye<376@!!OJ#8mHP-8hvF})A-u1+UP4*4Mn*T;XV!x zPKesH5XYQ4cHP;&H33!Unq9?_H)bV$V%k-+c{d5atMnIVmO>Y}6~3fgT#U-0l7wq- z0V$DmL4YKpl^MKC65)E;;6>IsLXT0J)%MkJ<2idozw}l|)C9p+PoU24Q$6}mJ!_xo zxk*=ojC-LoG_BHh^I2ah?i($ua*k_)UxMD|o{xUy*$l{!%s~R8Wa6)bs=|O*3?u#9 z5VVtf#^^v3c^7JRfEe5m7doS}XTF*$JqUOv7U*&c4iiIxm|_fS_DWt$02kqk+>4f) za>icp4hE zxEC)F(x$aU&>DIfkxGy_J~V;)jmXp1Ops~cz`fA*_JS;Sg>?bT_2J)ga?OX^Api0>v_AEtcHJ zbk!J6_PTf3UvWniWQo4#>RI;=oR@7`+^Z|J8fV(2YR+A5!_6Sd7viOuh5OT;EeZ{M z_g(mc7rtfcq{X2%czOtGhaz4oZlG3@t_ENE`g*B=4sZ>7UvviizDr*?tVZ>Folz|Q z?^>Ww|IP1x;)zl!wtv+1Y;P3mQT*&BPl{_N1$;_p@hcu!jh@l-^oel3NgwvpN9(7X zN5>i8xhLma@0}mU_Ri8r-ONVBli8@2wl>eJS!u74@df%j<=ysfuUil{k38AU-IbkQ zMB&Ts)FQp}o_nhr5~;K3YC5)cn%>2y)-itVtaaC-wMLf!I0}2JYlo?$fVA(=-QjU_lmHqtIN#n40>@B+j(G{VxakjNx@OT6hJjr-zKeZbz7uL#`=~6vYSgqC0wu}3nwY7X;y;}8z zDy2qd-M1C0p9eg(1IbtRdr!RWW~423I{utD6v_4dg6~}F`=$MCIN&O~Gb^otD}MmS zY8D!ald8*g7COmntyP4B@LsM`ZdBT9O0w9!^e4sUVIUbyY#uk<%26wtJ=k1dkK}!! zoRV%|){?okm0$=zo$7gN!aA{r!G3=-cfbH>AW)rFVAf+fRDR$sV%d^49=X~c^=l}hfkxR-9O$vaBg zpWh5$X56i#y-@J7mt0ADFV8o|wsTaF{fSsRnUD7qTb_z|+}W-rV%b{zIG?Dy{K1sq zMc+LkxKT=Od1`HGqnl2wRizCnUC9?Z-KNl3*-1yw!|~N_ykE?wj@!|$&yx#mGzv%l zPAhgA$OwT}HF>58dwad(PHZP|C`O})>*>gG>Ljx2>aND+^JXltF8kbD?Z8$*a7DMf zTUn{Lo5^J=(&1^{+h5DAR=P3wWv$X`N5LhT!ihJP%j~4YL;1{`5d6_yaU)W8i@i#q zBp+{eL?yAZn{9gf2U4_}jVY=0$$EIl9gGwYccbg6)ZwbXT0B2!9UpIJ&Nt87nOdOH z_B500<@N1$EWBB+=2PpD&dTLwZOxtAmm1!V5MHVI+=t0-=CtgV1%K;EXf*TQ^j>N& zy?a^r_uW#-pV+9KuNVCOrc~eFPh={g5Zz=Ri+5Mo3%On=6VLT?)j~Mt4T=5pW?)q+ zwuMe(t#X{*sPA<{p;S25NrWm(0M{b(d{*5Tb~fC>Wlx}M!(xUI>_XcJL!Y;MkUj#(n>Lj*IeZpPE+TPx=~#hsIS zb~opdnw9;v_HB4p?Q?&l-8=I3HdoV)=*nIsz86ZK2cqtR99ZwfR#RC|rF|3+Z0tmH zVstwoH=APeXs58=jtYTIsUUgtDJjr%ujF8&O&YeBa+mP)6Bsa>Bp zvguz9G^Mj>G16M;?xmyUPUdVi()G!$^X*DS=vQjB^K>>IQ>5-;dsC>pIUZck^!72P zi6e|@w7ilF?gT2Lw{zZT)>nMzLOxm+yuqYD+7i|y<&fB0_tbiO;cBa*)Pn6ywYsYo3}b7Ap9A$%FXDO1oKs1We1NW~6Wsik_a=&KrV9hD->poeAw! zK2^$}hgapoju<`YM!<&E&Uv_5J#j}*yBVM4mA8Yt;GJ42a2~1zPpV=zT|XD&oo;k9 zy5-yRHA9D)cBa+dEbO=X*;u;k-;0xwX{=U+^Qux=^T$sD-QZ@wkaIPfm2lN{l<1xZ zw<4fq6b^OQ!Qth8q2J#QtRHVh0xJ&LsCM zQnDqL)|(}BT*Xqe(unW2avQDHjHg{tva3htxC8RW&dFAmd*Uug=>or7H5>rV#b z#oo-k*G@*xj}K4%mA#GAYAf458J62S*@RGyZv}FpN@qXm-l!#yLm6)}xEJ2dN=J!U zWG7o$lk>i6<{;AaWrC4IUoVwd^_I7L2i}dq{zfr!-j2n+nUoNSNrkvi$w%GYC2d6uQ4N+dHpZfB}{+XviaxNQfwwvpTU6TAVbkk)TnBND5Top39caF_atf~V@LpQrM*c5)|q7O1&WrR1tt z4kob%dB7j8z;Our$4a+XC`a7kcxUIdxl`C}$0937K?sQR_G&zpKLHC?3&$&Mcg$a^ zRnxAn_Z;#ew^0hD0&%~uB=7iY2T>)ko!{xDHoM7oCfg4A$_>R84pcT$VmxtD+u86* z`v=)-FewH$j|-drXgwV3gTGf-{M*^n*v@Gru(}!DtNViC_)bHXYMbrj{;F^;VHrL3 zu1SftN@q{mIo}08?1=Fm^owXa5KqgUCd$Y$!r| z_R^KzqfW!0Qck2wzq^)?A18LY>+aR!Wj%5jj&_3S%JxAt-Q8(Po{eC_Q?0Z%+QC%r zFc~<5;JXa0`TOe!E^l}>bg68uBy-z6aqF}>PJinp+@byUd8)a-?cUs}egx~O4gWKJ zRX(h@H^lSxcCz0)4{q<);>pusD|p1~e4a3#?PbfyiAwS)eJt0HE6v07hA&-l7k0WW zZ`!-xjy3~ZXBD}7;ESH&bao-vZEt(x>8QUYY`G%c4R1LUI;pG(-Bb73W~*7QC!~%3 zP9C&8JB{wNz=X*a#kY1)$cDGpPxF;rYCXTYPP%AjBedyB%H3ip9%?D^kSnX{v!Uea z&MD(fUmz+MO5xR7JsWeaD&>=^oQ+-v&VBol(00GvKlPWA{p4vz3but)Uya7Bw!66> zOmqsJz=0T#r1GZ_UL9e3$9EcTB~nVp)jWqxI&U9rlutV6sq<}5DY3bVd9bq)r4=Ec zN_gcpwXe9NOc(Uaf_Jx4KCG6)E5Ss!inX^CXm&z<&-Qt|yAdihux3ZR z@t`Z12(NDz%Hc%6AIcpRp}bW*muGu9SA4G<$DD5TEADE>mx0ddYIqZoLJV_N^k(we zsJ9D#7R4CW;C>ft@a37er?k5HXbRe7I2c?h=DOXkoPn_-M1{y;2}$|1w-;Q6J(S*oeGn9c)x%m+$Of|wdpm~7!B}KJTapt| z&t7#Mrp8{lRu1K>Df##k8nm~!w_7_(tmpPyosDGUpthHlGrOVeY9}gXcdMc&9*3@e zd?I1tJvoT3XXE=3xm>V|ZcH6mZByy5U<}=~yL9`(Y;n;D(m)eg6w%r|fVy~7tk7p90 zy?)jo-_C5E)P#PZ-Z^VFo5||Vj&B2@2TaDD*LwMOC+1ygL5B=h%StcrmOZ&OC3K*q zw~xE=OtIZiGN-4>#wwg*av>j;1b>j^*2zJ2)JCY4JpDxCps2Hwy>dEkA7j zn(uHoyqfB#n@SjSa5oz}4{Z9&?bu$Wom`jK>+SV?O1#XRDVzIZyO7Sd_7WTNaeiMu zX$gBPD!+{C`1x=md7jvDZAH%Aa-j!Z*3SO4taaaCT8_10b_TDZK7h{aYu zB8t(F+qb_uYUA#wbBS$NCL-*eiCw>E-MybU@l+et_Le)kmdJTzX>*@=H@I8Od0K~o z^-yGm?B5e#zJ44G9q&Satfx=>*-kboR3!Njc9}bx+w&JTv(ZL(6FNsTQ*LD3P_Jv} z&75@Pl{fnR_>m~*#iHvlk*Rf>wM@;|&-g0ob}}4@<)e|n#%5G>Rrik$Q^y`p2ukar z_fqHwd!clsP-)cCVl>uMnyqZCB6UxE(bR5Y52n(|W(aUfqbCTIT&G=Sw{iCx`UDp6^kSJv8_uEc3=V*oRL-*=dH|XHmmegm1EY z>`#;{1u=PmKW^B{tx__7?B1z@nexG%R@!$UNB@nP z^Jb^vidDnej$aWn=kCo=HX7=e%9+?@uiY&Aqk+uXel~vEt@r)DsGL=@yZy|@K|dsI z2SdTmniR^P2h-AiHJ_KO(1VmLRE}WjC|wS9jU4eD9cAKsUu@Wib&4vtsSsd_}=WcOLv@A4FtM(;8dJU`9LXDdzL z8fe^bS8Cm;ySNeeYp7@T~bnl+0_axWBirh@Lz zz7F^T(Ny}h(5wcrV1~C2qI+RQsEe?;{V5F7#@TW7%ze;u6fo96@B_I5K@I9e|x-0^O1y&h>*4x%Bf*BhZ? zsCm#|i4^?q)M2kyT?_1oBIQ$ee;D0`%gcEDDh} zcqyHVq@ZVJ&*JNGakt^ENs-jHt9l#_b@w{L$w}z+sO^uhyA!Z35`oQBVo%O^mHuX< zzh2(k>0O?>x|`u#_uRd=32-(pt9#uExDJGLbA>>Zgvg~BJ-RIK!Yh6n-bW+(G0ZCdSWG$~izfN%#-_5w|}oxwp;^?;yT+8OaqA zY+oXpJW1^zZ3kD+wzl4}M3!^ABVsv%==bv5^y_Q%%Z*=%t6wO;gAYGZeiy%goGV{^ zMm_CY{Nm%Qh2ocS6yAydck%xo{(tp@tHF!77Zg{SU*Pg(Nx9m7ijD!ig;sEsn2Y59 zw?-uY&`E%86dgGUaF^=8Ue?3^&C6n~$Ct}T!<7ejvSPJ*L0gMggQqEL7hRH9{u2jl zPe0FW*WTcc5#40Y8a!zq666dl$qFP{du(iWbvtX#sqNiR4L~*n<9D*UPYQ* zI!sA^WMQU(>OF;8TBZYfKb649>B5}|Q)j4$W+bZ`jA)0pBNzwUk25F`O8l~ z!RY^bd6iWunhBfx2p(W*cjB2Ou9 zI9IP#drJ=exOPqGIdLf~u40yyDy|7YVy||CQVR<>!ij?gHK9yrp?j2$i@IZd5YZeW zPDsuZAnBGJFyw5P=CwLWIJ1Q@_n$|#*qcCEoFo%-T#_%-~!!hn}*DZ1kPG( zIC~&gk(~qknOd6-93L%ZX;h3m4C)ecpdmR~r(Vjb*}#x<`~kAi<($JKi7}SjY^zkw z5T;T@rxQ!1W{`rVzAkk@3Q8h}1ap+h9D&S$Pj<2^fDz1wNC704#-?Zv*};m)ZO14N zvf`+*qP`U`v(&%fitibthQ#1lsQ7+4xJVb#CC%{kd;nw z>O(}#g{i0qi&IxB^y*1xa0#MYizW*Y84NhilgKZ)@wyqNNR+50jV-Aho7!&>I9PoW_R++@1GJ9LI zV6rwB;mDygrv_)i=^2(Jhl5FbVAG~Gz0fkHM%B^_4PW%|jUleLs!hBGMhcF0qwRlV zF0T5FV7Ue(w_ZdztY&D$h7#jvF~|E~(C6{UGRDC+rjSkF(~rV`GS5-CKmDT1p>~s6 zHn>hr=l5n^=Q)^XZ)+Az-q96>TPYkI8Z$OD3l8tis`hYb%-+^4n5@r7|KjkOSBtY? zwQ@Ukki+7(E!aRky*&H)2B;(dwgqRCe zcqz0bYND)~X&}xmYa1jf-zV02y}WP}4q~|fJyXkkQ>#2KR|t{pH$gJ4+FhgC81g*+ z|J(ca7`d+VK1w!UBaxCQQ*uSixLMNPS#o#wK{AnPk0pxa$_y!rAsK1wU1~ayJG*yw zUYa|@-4U||nlwfXOc9`76-oXn+9D_rp#P*m3baLmz&iT$c-}#=`S!N+F)>`00@(X?8$Gu$3W;nrN z1J`fdj}^;!ZsRdRc5Rdkh}(c5UK@Kz2IV*5oCzUWAR_Xkh!=Pp6tJCgSi$@p0*-Rq z+=gEtzZ=#9nFKUU+N+UCbN7tT^g~;1CjAR(GT~k3E0N}g+GT#*kcsLpUq%jd;B#wC zd(wFqA=sUi=n5=3;Ev9d;29(aw z=fH6g6*Wiz1x+zzRZ=wB=A;K9$-)r(4cn2-;6O!qJ7Njhxs7tRk}G79)S$}nC{P@7 z!qKWD@Oo)togprROsC)`h*PN{lmw6i-FQ}Nj2tAMVH@75<;Nn1@LuNeLa_JlN>a8y zht--5nGTZJ3g4Izau!Wa3?ztRFG7&+Vm&HxGE^-_*mQ=M;MXER`jUu}2mFp%f~pg| z1i$*>&7SFY=fiAe;IN*xb71$NA~obgmDLNMr{k9=IUS#5YSs;O#B&UagSQiF#@`=d zJF%*n?WAvOrEh`PaZh!N5%QtQ(R!(GCpL2aW>9_1VOq}b;n7;)Iry20^{rR8_j!cB z^pc)kZjH;}Fo*JH@7>1_Tqw?0?*fi|-lZG}Hvyqu-@3?)!#vQ z=C!Ya>TZaX-$GYpsnHPW3|W7Wu!uBoQu!oKH=JtXw9>bBq!hwYR@>Zoe0#l}2|!mC zO%gTHDbaK{S5lyEjZ2D9dSOy_2Xx7 zDAEzE^Vc$5@lG@mLd{2NA(0x%ySL(nIGIV;RAoh^VS3+~&=^~%!NfpfU;IvF_jOOy z+DAOGYA^W6+jZCJcO!!`6o>GwvR;#FP?nW z7nUuMJHphUNy*G?wfz8be+}n?mY#fg4d^*4ed<#zU^zwqvGmt{*Ch|~cg&)EMphnX zkgaM%4f3B~g(U<0^!e1V&Lis_!bw?j2Gmf^Tm%70jhhPl2zq`Y`ACyR^Cmv z(}9gl^x_xarUEDKrS8>{gg;~?!RZ85LDjBql1}?7%rwdz%L`U2!cfRBB4|3Ivx|!( z?u|dX17L{sf*{hlk%{#84Uv9+1R}Mn86xew@2H!GwTJY&Y5Z!}O{0i-vX7VdkdYCh zzg7i&9j+G3d+%0pMX4RhUZGaC9f@V#$n z&cjTnF$V0syeipL5)6@5qf8+R&K^cpeB_ht=cV4oEGDI2DV&4%Bh>Hklcj!U8XJ+C zeK&U{mihia)u^kMZOhSKItQzoHDdop-Lo-0l$NK@srqub)hZUHksVnKLP)1gC-u=d zpp+(u(hlK%rqmi`2>5*aQ-gm)R_;eqiixnQ)pWH!=+4|H8$dqY6tddc0XLv!JI zu}4{phI$sk7<3|IFgyW$5UDv70sWa7K-DsH?Z-TYm!74Lc^A{qz8f#kmDSYO^`Q6d zB+I9*YD3M{3$%=xnS=4?do5oeaz%0Wy*?W>dzApvzcI=XGtz%<#`hnOfV5T3Al-k6 z|25^Boc?kE zPOs-v9%&)6Qk|mYP0EOzriip59J3$M2tAZlWh85@P@DuPgYlmTjQ`8X82{e^7+)%+ zynB#_Nj;E6jI>2Us6^Nzq*$KvWYZ0T*?EQI!73Z9{*S=wFGj}d&_B~huSJ`YGqlf{ z^6LFo1p#^~c$h}YQhy1>GhIx??Ik)Y`sy(@95WWro#u7=4E<-MYvw1+SUfE&4|5D# z)s7EYjdnS!nl&+=7(oK zRw6a~p6iNxJ#5HEwam=NqA)TZez-LWcSf!ngLVINmTY#cjvES$$!u8~l-Tlv;|K~^ zf!klxgTkW@MZkTv4g#9nT^o#U<{F6X3UMa>VxhpfBQl(usUIC2Vb^gT_ z9&*FJj~e`OT(JRgwb8yzFOJ1zlrJ_A|CEbgrwWXb1y9@><{^nv}v@G>bZz zIaSJ@>Gfc~^bj@Ba|qv(NdaL2U5sf5>vC?9N`fRsNegPpyMQImjRfLnA^{joV1-O2 z^lj4Oe42JYU|)ezLr0J-Gj@54r5xESJ$u7Oj+q+vNM=BoifOfr6#~c*ZmY$EA8KJi z)|`VF8pp#GUAV_`a&eJczZf~OP}Dk^JNOde(OU!1xS5Zgb+=cv9nc^~hAGMPFz>B_ z^deA@5eD*mZ0hORs=t|`dC5p6bcR*Z$Yh#-l;&ZSlYKS{9(*0Um|}>@k+TgR5jlQ@ zYc3rGx{n^3;0!??$Uz-0ICQAnDaQRQ@(Cd`8fFCB6V;{vQ>Tbw_xXepP1yV*VuI83 zs%DhZf!Yv827=Y+22>$Ik#ZK-1Gxm4EEn7|Jf;$LMc6oh?s9f&aIyJX1!@h^(~x5= zrkQ?I zi!W8Y09Sz7H_h~u-_x3apOkQ-T^TjLl#(jO#+UjrX~(d@Hf+dPybxv9O|YnrD%>ZO zQ`UA#q~9y|R?K}@?d zV2Vk(PjR769Y3t%LN&RHF$Gy$h0*srP<~RF9a1T^LJ{~UzWXHNp^F{e6Os7^x3mQP zgwr2ckr@Btt<>wl}073&|H)WP^ZiDQPkge(Q?Y(o< z@enDnNF*z7m{cwy3JD`6t9*Lnak8B$xmM`AdzsSQyjMyu3Fk+zy0A?6ES689dk-e3!Sk+jf{obtuPM=Uo!#Hvap#D zz;SSTsW+ip9qm^trgcZD)p^QV0BXB(-s$~`3Fy}%MCKk7J%Z_+Han6cKqk)@dn`~3 zx1L2}5CakPA$Ub9F8Ws){c*MeHE0unt<-nWeaXYB0wKq{p)K zAE~EjO8od#P^gXr zLLW2#?j zh|LRkzz)4If}w{}+ONE$2S;^-H?vpA90Uu~u|W+nL~_K<5UCgL5IsN1A&T3C)EKIm zgo8IBYXZ!&G@H=c%iwWuLJ`qwfW#WXBSB-5|AX;LP|bp~gzW&-h(d*e$cBD%s|Iq# zE+u^%!1q=Yq$#OGufI#$F*3-K+79LB=bbAG_kF)n(GDKkc_IX(X)k^DcoeZRh^sF6 z7;*#+K&qf%7i+*Ox(41=7%2Mi-7kO-1~R-s&Lf&B{6?_eoC+HVoLTT6xA*qaE*$$w zxH?*EY;`ijWKr&h2JQV|wE3Hl^>=VJbrp77c9Pk<$?z zhT@hknGHta#~aKUzK+FHnrXMww^$4|{l_()YdY0UR%#=O`{ZsZ_a`|hFY=kvy_~E8 z=~*OZFr2CFoV3wg+fyh;U%TiPCyCytdu;<2e!a`YAUND?qtr3A&sZ&)cQKINq5BDY z=tkI`)EFPQzB?H|JITq2JHFHk9sc;TmUrCo)!BPM%vH9Cyu5TC6%hK`-`9F!%0CTHLR5 zk8kX4l>K6J-D#t}x8Ss)xfs;kXiz@vhbetg`uKKGPck3+{~UBUk!#rNe_h`+%%mAc zxAOudk9n{=4Sxsm_vyM#_EXw0zSn2n+tTmSiaAsI<|qq?Z^rk2HKu1SIejlN?#y>l zT97OfeUX4SX9y~ufF$*RjxIzAHb`0|dH`_pqGSm3M*&q$G)&_EVNxWX`(_Lw84_Q+ zh`SsPB9J!4EWnY7YnIM9AkHy9-IREr^cdyG9fIv%R#YCU|6Vy1D13blzsxYFb<)}2Q6J6IadUXyCG1gV2ZF8 z(Ou;h5?VhbFQbTPip&Yp!G%06;8xEf6zUE%9c(08+ zc-9d=J}m4t|0%N9#POpsWmYFMhZZ)If;cix>P7|3uZ~Sp7D5vMHV!VU7KQS1Iv8fRmgFrN*J%6C1G)3uA1ukft?!x`lvXAk?CL&88w}cWXl=#?EN5 z!-;PYNOP(g&cH*UEvd3c+aQafb+{7;xDptn4eYhp`&kOedTLL@BMCNKQ(F!6vq4FJ zL=+-*a}u&Zl7q8XT(g$n!VVtf<=&kuqdb(=6}G2#Ti0ywxl{r?9s45VQ~*Y7w48@? zRIMolkff34GeI69{w@XD72g;)eSv#z+~KzfetcM1CW{d&c9)4GRB6Y{TJYLk7_US? z4t3?*H48&^i5G?po~v8v$QJ{<-eV2Ri))Xsx*2AuZcP%A=VKOw8MsdD=YdEbQ#u;8 zbD0+|D(_Abl}`zt%3Zn^F7e+7BGY3|MwzkqCW+=3$JlnP(HINNph4>^=>D6N9E~}l zNgHAtwQwJy(jZJbFF0q6-klZY_xkLvgTFJ$A&EP*)#jxR?KoCCrp#+&pD-HF&iJrm z=3=lk%i6&W^Wlmgeq{pUhlrrTkz5t`E;vM@__~4eta&ZZ`Z#w{PcffP{{*Bw5sB8% zzI4PkP3$r(_$lJ>vIS1;NI#K&2&R;9>xMvE8tn}!_<{Zuk#Bqv?S$mnXZnD)wvoQY zK&<~VvZ`GsW;AJ%+D}HT<<8B&i*#rx7U*wZ)$^bbVixGA8a3TUN56_ee(*}Q8j7sc z51#no5dTv_w>p3J@mjJMMiMlRHit&MZMK6SRrEEf@mBN4liF(j{+1?jLwQEB*jZ)! zig_#I40{+?kjWHxR9sHGZdM@++aWp>^7bYqcShJgDR6o;N`FBM`%LM7jk3O3IRERv z7=yWcI)3LvUTUP*iJYF!X{lK%Bc_+X#Hwpl9>1M!NR3h7nitfy#4MyAPUJLU@G?32 z=?R_4zx5k2#9}Rz50n`I31A-93vHm^^GO=NBg*CFdM^p^aF|{>T$qS2w$_3uGWgJEH znL^_P5Xa*ODk0{GdVwDVB`tzSTCL_>H_@Z>Cq_W9Czoq_eCVK95qAZ3OY*p%qiPzVs0E zV(1i|BX|ML8HRMzN7yJ$#4^;N%Bj2@J_o{jBF?AwldTIVoh)42(Fq zn4ug-0+wXH2=*d+P627yy_e^N<0Ogp=Ta(wksu_{JOaa%DV_+=uYbk6VYC#mPytD( zLmql@%xu-$Ni|Z~2k%7xI-fLNrp8;5vd zm|6A;<BYHfr@cQ8-%(wF93b5OJWMR#jYe z(0R~NtK1Az31JJtTNTOPle9bW4@tAgt|O>|b}A7QDvfQ)!DG&wOK?9eFs8Z!D8Tdr zLEr&K4YgQ3x<_~ofJA-0>Z5KQQInd<*&B^Ut#JiWk{C7p`gP3YyDRjbx~gBlo#nEs zUH$&?p7-Unhrbu-5B-`=(fDSYNro%H%xn4<&lfZttO*@BXm|a}6@-$zf_0B*PwHox zBKo9fi-@NtBlaNhYZKcDt!=eU^M7T@>+onr94a6g_NWj))Yv2?gu56(SyhJuqF%a@ zfTkSDlq0c&Jx)0i!=g?(k|{?rOVJDU!0x zbOmO8L*IPwlSh?8^Ol4YRg{hz`bNo(Oz4|~$d7*g$W(qb8TrwrIPp;>yK5v!mlDHB zgdh$YS0XeF`(@qSM7UesRr@y9^-0}O#0{(KGsU^p2;)A{RX__KtXqnZE{!ToBB_UL zN{c=W4kCK>hbV03C4HX(jZbRkhpXkOxgfCRv%-gycQwpFf zeu*%fV6%h_R zR%W0YcZ#_3gV-W(yRwAXGK1s_s*;c=n5XE#xZkzhK~*4$ru zfz#7VO38;fz1(|gINrUqKF+=L`xo@RwDO_dOO2V(TH;CJr%U`mcNkRI@yb+hLO%tU zsRGg&L1P8aFXTTWrB+Z)RI>RkQX(tI-FXioFW^B1%~}I-#ljYJYmcCXFIXiIm0-t* z0uf3!#3qHauD=^J5xD_gy;8+V>PFTWRuD;p{e6{gQRwN&6Ad8YFvjl_0rZgG{eTpW ztlx)%t%hJ6$e)AI8qn}T3yNGsY?X{6NP(scK$%C$n3}bO@Zc&V8`OgAlqDu&G z72p#(!)USAl{y6dCkDJ3h}0cA&zmrB0E5Q)tJtCg`cBdpW)tvA2I32(@W(hw`vD#t z(xoOt3j2np6u&oTVxzLl@HjMK85)fb3pPo=s25f57^yL3ULBJ(Ei}dkH8K`TtOXFj3xqZ9DHt1a`Uhdn=IUMFno2)9nC^&Y*6Fi@!GgMtA)t; zu*Sh;dGsw_^E#aaRF3KEVB5+k?`xh$$;e@%s?(mXy)R>hh)iM1+Z-H+4+nDur4=V_ zHK8qkDmd2=ElU^_X&V?e)se;=u=-Js>LR}+PD8{mZBk5eSmbCvM8%KfCJ-|OMl{)~ zk09w_TZpNF1zzaE!c3o$h%5;%1~2GvOpz5FAL&h9fQ^8vINZS{5fFHQS9Mk{IMTyQ zq1m2iFe<(mfud?0KHVygNbF=b#cuaV11ZLVI3k5|f>B0pL3VHu=7Qw_?({zO@EjQ5 zq;im-J11skOm_E4f z4=0?hM!82u-s*#g9bumTE3Q9U`W~_8iw>t%+NltQ_@?M8= zFJ&H~XyxdU#)jpe25;Cu65_aI)3J=AwVGFG4An(EnV2D#(Cp^IzLHZy`I2gS;=H_4_Gd8Gk@c4?mq-mo#*d}EmG&z{#VD_4# zxr6SS7DnU48V8eW6WG2jWF`V`9DHuur|h^#iG|o?;f{mh$^p(d!@?b;c1&*vGk2#A zpBcy8buxdo)`IH}uJ@Gj3^EwfRNw}Wpm^law9eX(vi~r-)oj+oD~pReJ3Fa*6%vD5 zV{?)6cA?nf^74-_UR+sRzWmCiFU*nVgWcS)I?b8>+5Qb#b3CXd#^L9pXYoZH!sDpy zv0~lzvl)E-3XnPxNWGc&t!tgRp$B1~GU2Tt6P{MyOZ;>dIZnVgbBLTrU_#%5PnUe|Uo-K5NX_5PlF|&;PZh+q z#bJfJ9&?t+dIJ7YWHo`hP6OrJ2BqawjpD+afQKyIPKaS8={33XW*Z>mO>p9@-DABM znPF6Gl>maXqMt=7dTI(ufpmryA6~XqSy_cu1WFHZX<)x+a!3-rQ(w_l+>Oxm4*ANt zz~&R;B10lE)+Exb)7FT9_oQZ2FjB-mQ_vw1oRUhgS`58oJs4gj5Rf08ZdDPA2)atT z<5MxybG+ra!}@X}baH4J*Bc~6s+Y5EUjzsiKCO}qj~f#7(tr_GX&8ie#PDhXUN#XK zC@*ONLAu3+ts+t=l4mUqmAq}4Ym~&A1{@XeKMKu>o7wsEHUXSJpMtxz z)fV-aX4>~ICCrFOGUCi%X5h*dN;W|9e>l&U+t!_lOi$#UMB>ahBB9#ke3j?VOR_tE zo=4x|te}-m`6=RlS$iy~L?|`ItQ$0mv*b9NJ_XDK!Nd2P<*AVC!f&+=d@cG$V~brQ zKp{f;)|-fC9+eX%-9fpZd}mi8Az_}Igs>G`wYbBf7XkVN5~6#g5J8BM5WiV#TMo9V zYeaQoNJJL3GQ`)Iry9huG@gcFoxw>ZXdjWDkeN)Omd|8h2`HWD?f3!mRnL$Ej;p_N z&{y90cVi8%;hu@T1P+fkLQUa>hOwL^YAfMfm|h6*2~BR&>j zA-P1aGHRDddlywe0SYg$4oBEaaQ46aFp*dj_QPSV(M%c`1PbolvObJ5DJp=~CwP~- zg{xL+^oU~+kR4|+*u*Xh-}X)R#!oJz-OJsc<7K~^l#6OkBAnetlkDW#%CU(jI&Bxj zdZNe#($s`eVLA!_(YE@D;b8MxJ*yeL!Kc0Mxt5)g-umHs@Nk~% z6O5t$L6%f*75ofIy`(Qc1CV569X#poNB|Ni6>3M(K!M8-wCFVDP5i6duv05hwKyJBiRAWwUBFj=*gxAhtZ9(K`% zO@#9=*Ba^B9BjnH^a^?}48msaGv^dqb$9toa%R_h*Ou?v*1KJp5+x~?$1>Kc8iO{! zT!1(IiuG1HWS49C^h^#sy#)mtz_Op5xh~G$$9o%9QsKcT{Yq-Z-%DOtT#!HKW|WUQ zH%1N0-GEk}w;%{b`1fOg9C$MreOJ|j45mjG`~1y zO#QM_L`)geZN7=5ri`hdo-(FAnF0PQ9;<6#t=w6Ya|XpMU@lAx%E(Ax?ScETmhfFU?N*a=}1c}{*tOWxB;V&zvivc8aiM#jS`pFyD9%Syl_11&T z^_%x_0h_re!l=z8R&T!X)Aw#6BJ|(DQ#28l@Kb#v^q~CO=SfP8fpOozxsoajl)DploSAH0y6;Vc7boT z_(C7Ly9v?P3>0mf5NALx4Qfx%;t77siJ(f;j_It`NvNhfH*=$QKa~$3&CJcuHmKlP z@1CC(Y65%`6oWP7p$rfN2%;mX@v>EAflR#aftnCi z{HovB3_y4U6(YO>B8+AW8GucSzCn5f-zo^`X0MBCkXy{&pjz!&T<1}3T|H7T003y# zGFbOM)xu;iXi$Br0JNmRBckNEkJ006xQUd$H2^$Azc%S7Y|Y??iyK<@<2u~Si)pbkCe)y zUl~GAiQP4RqGuLvh(R)PWLwDl6eWW>Lsv5P{WzG{4-}s&^l6r&4XTU_e<%9>a5^#c=%p$nreePmm=5y8?o2aBvl>>jy zyon*u7TIDUlj2Qc%NsMGrCuq{c%Rd+y9QwxkeHGvxHyu>y3#UlgAoh;%eBqrrKIW` zcS1|7XCJ68-=Lr^GKhu_&mk1+HWbyi1%Y~K*Omh7*}Gc5AvX^RT?(&fxnJv8$PeX* zK5;^)jHVYV(a)68G|Z$p$B0#(GMbtno-&%?N}+?|t!~O_Zt^(v;UUd=(WlUq(VQ}x z%xEzOIb}4(s_IMUvkzo(_W|LS8U?fuiEKX}sKfy}Zi z#3ierZhOC5t4PywK+C|X)R#onB;<@0J}cR7HzTjUOr^e4sqe>J>I(@7I{_CmmC(<~ zi+UekdMJi_+e=?gY$J&jN%VLJB75C>Ie@cY+S|s8`azKPD$(Ryw@I-ORn$S!KT&D0 zI}tQ?G*(elRrCc`sUx!x5WNW{Ta0B?UOlp~+XY;1=dy z4iw+Hbz}n{u@G*0{#T3?E}m0}kQ&EupK5)l=l@je`@<}Zi`|@Y@bpsJPS1bo!Z|(v zVMRh*%E$ZscV#30eLUI7e;r3Qvhq7xHu9f8R!mJI>(`36Ld>cIsM~)&waSm0yhNT7U z-e7V;8HC$jqg6#7WY-h&;_~IobTfE^?2^crmcWrE6G~!Rg_l=%Zsdu>Gm!U3971E8OGwE|rEqe?><#sN!S0~0>wY~aEk-{ov2SD@sv}T1AxSzsw@rPO+;9ReFT|;WyK8=mSBk8gt(0f+%CQT*UN#2nrSKk zuqD2ZmMb}mA|<;hggOl(Oq5^(O=ipWtt_5UQkXxu?ew~{F%bBPvIE^H*EPC;x=pCt z8Rs3w5Ry4dGX7FVmEp1ql4I)%*!7WTtZ-jrBMC7BFt3mp<3<+SnDp-h-m{S_g0y{5 zMiatVLqJoxT9v^BOeF?eN=!E*STjyHB96o`-H4_e5t0koQpt~CBXYUL=U*DjZ{k^0 z-+E%}6DD!t2!0c<|B2=nPe?eE_nR=mdpa+HPg$Vw0Xkd70?h6Teyv#C&u)1S_sL;| zOL2&1U5YFORjG3*_4$V!Z76|gL#K`(KE$}6*?OABp_Q}RJB4c$`HMqzxgg^(ns&AT zsr1^ZLx-{!yf;tHN@$-TXrI!8hTugv@VTNfw-(i5GN`4$3$RAI_0 z-rwmwt=fsaxW_Wl?G?M}BZ9FFAW2;fd(}^acj^wiWZTgDl4x?X<_I~G7u~uus z)d9gA^v7FIbWVf_XaasTvkU&SvX3_>5uGTgXOO{wGKYi!c}BeP5vGXVWXM)NlP_n( z5M4%#vVwl`G$N585J#pAcDB5K>&Z@HD_ho&hda+v?vM;c%xs0K(BaPEf}d+`0^Z4; zY@>>?LJ64MSN89e{_?p)hkl`RGN{&D%?x5}*9!Q&-#OlD78hPYz6|{SjQdeUNys3S zY5^%TFu>Hud~?6^Os&;)p&Vu?`<<5~Um_q%e+|o}bCP8?Rv%RD&$u5Vl{?SUpobWt zEsRY$3)<|I#jlj;xMB8g2R>okE)fhb)oY#kbyF& z$`|^ zN`{O3ozql}ca_iKppcxvY!u~IG zo@!)w^dnZ>vtb)4dUi7ujHrzNgG@hJt`UVljp(&lNbj|?zlR@q6Rvn!=iA{&&Sjct$br4A{4@VYEwOaV6Xe$U?Ik()iKJKao zti?9k+VcG-pqa_ebsZR?h3>MFJTC2DGwTINRAp29C~^csDjMoSdI zuGKkVlM`cjA|Fs11Pl?d7H??Dt8!u}?Kh$v1{g$D$<@l5Lb>RtA}@96;||2*s#UIZ z%7KqTK83I@WlTPBz-e^!ntcBajBTk3)Nefz{`~$u x+61!5w}enPVTe6ZG_3b0=k-(iHIo+tO<~KFy9=G*oo9=cj7>;1ZmnGE{{fdzRowsp literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/.doctrees/chapter3.doctree b/doc/LectureNotes/_build/.doctrees/chapter3.doctree new file mode 100644 index 0000000000000000000000000000000000000000..00f1ab144e2302e28ebfca395c001c7c2651e602 GIT binary patch literal 144166 zcmeIb3z#I=RVFB@)vr{m)oNL7TW!l>sVr5s>Xlh<*)3b$EsgbR+uf~JnSK$?)~KZZo2s<`Y*Vn)hPS5 zjd8bBs?|$gYbQ8RsxNoeyjpwb6FVc{w6nc492{)A>#cgHx$Nx(H=)FG*{_zGUTx>f zPH>otxBY5sJgmRJ#FJ;$w;oTuGZ ztL~ihR-0a{MJ-S*RJ!$ow?WV(mi?9;1}i$dE7;d?0bP5id}pWV206FgZu-Se8`u#x zd;C(1Am6`SMTdBa-+OEB8vk{m)M==PqPu*p>8((iPO!i2uF|KS@*TmgRlnx#JahEH zV4tAO`zM}ldCk_uTDjuYT3&79g6D!#wbhAd*uUP>p4+U^Ge|b^G^o^RdS~kZYhtCYbb6aEUUa{e>dLv?-4uVo12oE+9 ztW(|_9Ek$UiyyOmkYi~12AWKlZ!YgCpEz3{Ig4>EzYT*~z84(g;Q#mG|3moy0~pMG zfTPp&cY>|gkSaT!@^HCO9(m?X50+1q9|HK}1b!0WRAW@HafsI_-^S%o3DObqB5*F0 zJ*S0{>r~yQbKR|WJg4L>*Vh{Lme0hH3DL0>1ts;%9|W|CBqm7j0HkjgxIG1S9TPw; z{vwnM!R2kZYYlLgv+g-%_qyj)>-B41$#L5b1bqbnLr$G#7eeT)m%%zzuvTxbxz*|> z-q&(|?Yh@&J8d)u25$Nr&d}vzy;^E*uHm1p#qFyX%;%wT=UF0$<2sFcb+cAq^W7={ zX}C2~DUM6PaW%W!0Tmyxy%rSR)baoJk`xyz1c3;SL-zZSHh5?A?ip2 z87iUaE<-6$tI)m98q{-j-1$JU?fN9DtOBD?%d_|F1HiIJP`B~7RIho?b>9VLTJ^DI zK!K5X`q@$EL+2b&;dHr1Jf4T4oK>v{p8-kek2 zVCsn%I2vawRU@!{jWvQ<2@Gkd<$Fy|jB=(>d%cQD3Ng?dI&R%{9tmU(0{K=Ukc=h} z%40z)iO|qbS3X9N?qrAZ8+qy`2Fs%Z3!VmO2950wi}wyF{2C4eH?=zq{?XTr`jOHQ zeC@y-9TTsssUl{4DYj3AGrpWlgU?+kLyxbyo4Im*-NS5%36f8UMLu;pe$83+VB%pa zAuAf{y^Vj}W=YIIe3HQYm2<1B^`_r0uQ^3bx+MowTFc$+ISWwt4lzB=Q^MGQb>X$f zo#!xV`_-z`^0euJ&6@zb=vC|MY7RLdaM2_{y#v{PNY{2YO&%?9YH*@xuw>C%A6RIL z5r*4qGBQfMUMmQq^y)1v?-X-fTv+~B7%8wK!Fhr&TxT7;3%12dikn`}vfso4_c}~5 zm`5nuXx58WZ;cC26^7|53@q44YiI~Y<7&AL!?0NIaH*>|OMcC5GI?mhj)iH6<*MU0 z&NLD2 zR@|)|)my7$7#q;7&6Q3SBUHmGoyvN}CTwLPmg5fA+HI$UF~k~%RxmDB5i4xH+ZJ}Q zTEvjTWGkMKxv=W+vD~2kBHEcRaroLqb;kJEh27qyB~b}@f)~R@0@u4K(R*@}j2xTX7XJjA?*mxFhfaz1X2ZuSUQA!i$pYV7y;Eaa zgVar^%+v%kO2#6a3746_C?v^kiS{K%VbDml$_;ywX;^!{V5T5&ybTN)e6&>;|4cT$!mDE5@z#J z zK-RV2CZu6K6YbHvPmZ>h4+Dbw%HjXaPs183xosCS2Kj1YBQfqg3ZtPWW{y#y4g!{- z?-dx*_i{GdVb`QJ>z2#MD|dlSb>(d_<85V*L|IXTGXBlhPhI#QgTtmLModo8*HYBU z(^7d?1n%TXhny|r;iMzcVe`(^E@Vc-bywAGDoj7CxD$vlLeik!;(wVL0G` zKbp~iN8OEfGC!V;-qHx1Rf{ t`$i&rdi1_P?jtbS56j31O1YrWAc$YN7`F;CzSr z-wAFlx7VsW!NHBS>cg#uOS_B*8dy|$&Fda*qf{2DQC+iW2F$ta)wo9ESE6+T6TX?1 ztRppSpUHNDkz*HR+^_}5-nNg>r)6RL91Jkqx64b#YJK^d&WzucyIk~E{o0mW#ZJqK z?c8N5S6sn{kn@0Z(s3?Vy_NPQS*qD|H@CN%&E|G4j{;-#r;xjRnLSP7Auk>Z4*M=1 z^Yl2Gy9^6s%Vn)Nmz(4gxDvh#`xjF2`U6Z3F`MLGgd`yqw?|eIcb!g*v z=WL~dN|kF=6SDY?B#R@@+;=wI)f8hTJ(i2#i4w|~W4mH8D~{%zxCO_&iR4IRei;u& z7A!2-Ps{YigLQ>JD}EE}_0Uj+DMs(G2RF9vV;ij1Srg_lcK67bTCay?Lkp9cbi-@d z@y7dR6B~D< z4=Gk1B2$v~7OP}C5+a=BITVw`@nPGB2Cl;72{PJISu^En7nQZaEb0w-A6gDWwxApY zcU(Xp97Z-Qnxcz+f&_n@hFaTlFAq53ek#&lZwsr?;any`VHVa5+f9`elJrSt9}U@E zHee%(#mb72?O1S;1~+@o)lhn+dgvQxZL)fTu+BuVG(|Y3TzJ6t8i6QshGz@M$xpG->6ovjq!TqJQ;VMSdmde*uiAGDxxO_ zb*T=NY3q}@)!{h_Q5dCaw28?k#1J;&UgBP*FlxnNdqX8}4W5>E6FW3Cv->S3OyoX) zYgeLc+@4SfeDKE+vD0SEL;Ao80XAPs9%ZK5S0iopo-OCaY$?ECv*qywCeaW}ES8L# zu7<*BGyRo`*8W>eqhhyc9U#{mF|~$+dS?UJqI(3PL{K5P;2k|=MC-0+Q5zIWq4x>| z3Q5_km`UFIGa~Po-k7p3i50v-5wx&e@5x9g4lx;3z6rWVdQ%&687&suMg%4n$lVv* zV!bWa(_>g@$Jf&1_Qmw8-?3y5rh(rBS;}OweCpF(7t6;%X@t^B`51ZcX5s-UhX)B-Krhf{BZD))<;|D$C^+_i$Y=q&l{0P3A&nKNp1o?$)Q`1hs zUWwR5=er%)XQvQctSzzJ>|iarf(`Q0INvmqp7jc+I+6y1vOqBuy9Fze5_HjwnQJO1}Ea(pXILuXHyi z(>@!)xA**d?&rfrwjdQV8xBei(WtDwOeU>9|MxTr${B%tf_O9?|7tk(Q_F_5#jFs7 z)Xa*p3*@g!D=->(eO9zVs7#LN4p=xe(bpzN{hiE)hkR1$#9l%Woajec#Kpb$dM!0`S+WH>a179Gb_ z!}|{-?e$g|7lpzQpfH8;P(O4~F}3&AR~?BevnhTK)K0c3!dJ(E?c01l#LLynd(Wc%|Ih@&2f*wfd{zNmAl*-@EYhhDV{8SBAxflgkF-I&K&tLnqU45|gRP z(6XfMj);XA_dJm`3P%slgi(VqLJ`{oFB6@QL~t!B-eIo=znDu0u|}vpaLc{grFV2jr}BP1hefVitRbAc$mC_)NVk;P1D4ZZV~@7hl;zaR3$rU{&EG_z?>Y58*}U`S zn?h?{pv5NNNfw;ehk$6<_g^=eRQp5O6FO8roBHbLo<1FcU6H?eT*C$rHt;M~x*#B< zr-+6cPOB}nF4NbsL8f@@!ITO}5H@vrlknr25q^KD2{l^CS2k4B4#K!tETvPdTKE0d zv_Yu`WLNg|lbMnCEUg!_n;|qZW!2CNY5%4r!S_Hp2a+7pk8pyCbM{Ai@j#3{A%5|ti#azo9k z0wy-A;{4AmO)iDRtI(N+=Dj&g5l3*Cxjo<;t2Sh0h_BAVm^lQ{Az6ZHQd>$tgeBTq zvE&*hpTN>hFu<$WqZ}T#c>f`=OY-79-4g4`7w3f;Cupik?Z4voys;#DVw(kvkr3jL z0Fp?wH$!Q}NrvZWp#cdj`K{$ntA&$r?1&+bylHBcZI44zky8QzXSDnw(M2sRe4S3i z@yPv-EEly^zG*I3R#;2%@n8^yc_Un>yooavve3Bm429H*W0L5nN=G2o+F~C^O49}6 z4VdzOj0kNvm(~;8f(4A38L!Q8`q@Z(iAsR-COrA^B{3ip7^VT3b2uH<6%6oMf&eLN z+0g$Tx%al0jikUje3Hlp;`8y!uk?j7j?8sM`SEPHMk7a7E!s$>&-0v31+n&rY^wla zGW>ge5h{r{rf8J=l?OF;yQ*#>L2nJNm^-d*7QYOa&g0GcnmU&nM(&6suuu`#pw{?Q z9}Uehl++@W6dW|-fggh$w0 zZd33W+hc9G2%lyWLJ?07?|$HwR1MS$`1!v^1hSj*U$iTKyp}S>1bpAE{rVNYPbq%^ z3{(COq~m7Qe~`9@4Q0w?rk@Ay z4w@OA>}@)K6lVGg*dz=3B+8PtZYIhEJ2tN{cuZoQBmi7P&^&8DZ## z)~jIi-Bp%oR(_Z``FQ0Y^u^G~ob%*oe=LfUIkfM~>En6%yuZ5U!oq|(U2d*zoq_8Q z@rL2J=0_bq)Z#^LWL!F725(4W%LetIWi_au`(l&=+Mrf7QH)sra_zE?3~7`w6Uzs< zFxJ>fIjm|_%DjnA&2Li*D4ZgwwRpS+Wd}7He0FU3;ek&19!Z2Ivw4C&<8_O9;b_nHf4qg0ks=?Kc>=@ym<}1*$q;q9~ zXnedf*B7bD7Wt~=H9-th)19_yI+oU<$lZI}PA>m|L+KhQRRiU1z`ra~PZW^mNG5Gb zoS8-P7!c0H%)I4!A#;KbnwI-rIFTu&%^ZIu>SFts>g&#*-aj(NU5i52s7PWe2D4W&Nx{kw&`2Gxd0N>qmrp$TW+flr z$Eql;y7={otb9Mum(4TsitTUGbU6L()mDupsw!P172gyIwJH6O<j)gzbo){hStOu*^mG|G8;(T)lx90mdP>1joP8R0Y>@?3d)p`6wIJ? zoMVm`Ni4b|L(nK{+7GdfMWc_4+d~LGXGi^Mta;KE?1mv*Z=^$UMskg}LZxvb5Y5D9 zLAh&)9k%DftveL1F$SmQ{pk!iP1DwyGzU5Qk%;26#YS7o><@WfgGj!zwW^fA5tNa3 zyY6QqPW(4CoN@eHx!@NU*vYNF5Mdw&bxvWtlvzY z7|x8acUdNT4MpqcWYjc94&5Kx8xxIpXNE>PX+EA{x@brz7E4Cn5R~^r2?cm?Qx+F8(=2Eom{IHsS1( z_u=Gxc=AyZL#_COM;?VAUCXCWoXdRVl)&F`n~IrUTPCSC(O4!F7>1$-Z^D^?b9vRoI5#*7T)2%Zr_tAWRJ6r zQ9U4L=rWQ{uDMJ3lLg_7;V%j&DajNINruyk9~6a`4YzT%-pO=5c~0((wWm!imXDC9 z-;KzVwgH<{u!z}2e`V6_KSkQvJ?Er}z5V19=Ln@E(C0leC@9s&zGMS-PPCa`~# z8L+mwF9xN>$gz~cZBoqt8yQ;(Io(HPD_|MK-gBWwe5?BZXJKDCq1Bn2(;03FZz^+w zA~OU!Js2iHj%oV8B%+hENPI3FKpmh6qXo6!{Pd1TFI111*xl8;Oec` zG^VquwZ%_iAB$y#LD*?{$15d<5ewQB$;s#-kknYYA+g>HRh4kqVPmsh#^r`WWiZz{ z&Uuqw6IE(M`mN=*oiW5b(q`aLLu9C?a4dPxxvd6vYuI*=3mC&w=eA;U3m*}C}}$4O8;;3TO)@WLy`T}R?cnV5(>0~x+peOa~^RTFoZCOG+5e!TzQu|$t`tM z>ntN7W)pjk#$EIxDHXxPoW&@N;iCT`9VZ5g>I$ataSs{VV@Nzf6bY{*6K6t(8i1z; zcz_IYdJ;1Hc6e$a+zkvb`pu=dw2dhOGUUekVp2`0`>Y5d`$$B?Xa=C+980?a^=*;% zde0PJr)CNP44Wwq<5-_2Y#Nr)FE%tRI!v@~F(0LyL@HzLjnrOjD*XGIAvnxbatfWH zp`N~ujVNj&mFV$(SMZ$n#zgp^W`=Og>#IgY8ZXC!qj4pt5y18}8C8Rg)B1QnpBd7( zk`C7p;Lph*ACxLz*aq4gb9DY)W`JVvAU3;ba9ckoqhU}!t)u!+GeamHyB(uBD?VXr zn0G6YjJ$PQDBNMOely2)?*Vg_vU@$ck6)IoAtc`PK)_-b2zI>7_RU1#Etw&ZZXAz9 zEfx*o__t)#51Yo0z&iTcM0g@Igda?4Txl4lu3|$xajw)f^Fw4C;U>W}#+(LQU)5~z zscGwGIX~lm);@@llrzPlN)4bLCiGH#5qR6tB*~NlQ!zYtGtC^8! zv>)?ZL#e-dHiXlX^>hX{zXo_F%?*OjTS)F~>kkdc?qzH}YplbTFp_onlgT4iSWrs|@lvA9p)=p>&E(F#Tm0?$N#u`xvyJEK-Dni9Rfha#7;iGO#&9pZ9Pe$77 zZ6bMUsVRVAOU-u=Ks}9J2WqAd0;M&jePRf@*I8(Z)^*h^iL6 z9ng)nu9wE#u@wTEU6;aQPPa!*COM~DdPB#Jcgq` z^|ehLD|mtJ*9hGj|Db3V{{|ZUDTXg;vtLB+uPUzoV^&4m$N#e@*hO&6&Y%HQPBi9x z;BGXVkC0S~;eUbiEhnSdc~UzT-#t_tgWpcO1~sOcigcg?k>30OLC(iLu~p>9tKskz z?;mPwxq}RMh{W2m4*N`D*k;*GwlJjSr9xyya^$@Wx_1t%`kXX6W1z;U*22*r#Mf za*7XyXz8z{P%&@aXl4l9rt!Q6yzx32@I8`9_xT3VK@?w z*oF<{C80afTXd#ho97t4b1RhWkt_$h=7{ja)+j8hwjQ{TZ~Pw)WJO6!m`3$C9lx)9Umk^GF_}P0Cl}X*UD4< zHJ|OA5d`k?atCpBo7g2|duj&7)x7=Kg)D|f6N8Kzga~~s!sxc(5lfk|zSu;Ayw8d@ z2s0D%AB^DMd%aatwv8YPvu)nejpsGYyO*&cY0>whMG&&|fjriiV$yKyv!aa%X}RLv1874H;(?kO1pYImxChn98Y0=W zHwZ>YX}X9^57Jx?Y+o(C^|li<1!c7pz7uu=7Lj4x3A={tn&&lm@?`>%ABd*N6)|sO z-Z!RZ=9|Ot7%#wY!n7i2T`2}OeUynGPa7o7#0?VE>k=VLSyRq zf#mv+KmCCZo@c-z-2vKwJ=XG?!Z2Z1MS`!T;h=1UJ)n-pvFs5naBI$7fA_a~zdcgB z_YBujGo0v}&2Vq+$}1Ys3560-sXRkAEp7%O>m#}k=46J0xm;AzXTDE_G}zUAS;!29 zy^OUql7}ynfgW%%^jK!#$2=oy5MrN_QPSc^#3@e|VsG=*BRnmqkhr-fJVy9sj?=ef2D~ZI zvX@Fe1(gi}PitL%Z)Tt$u+R+>qu&oNW-9y zN;Wjp67K3NY=cRf2CsBa7Ao{M>NT588TG;KYn!e1Qmr^%tK$Olo#4P}Tv&S+zqPy3 zz$0|uTIIihO@iC0z!No?&(>17UhVd(kIWNpb*WgbFJIHzS^W7uH{As183HO1hrZUp zRu)p7HFEcpjf`Ed>$qW;5>VkRExo~u;YUXv&fyHG^WcLIIv3n)l*W=m($PD%DR~>A zWARaD)4=%dG_hUA>pDdQoR=tU%=9^$ODKoseHmx{FhBrQXU%K+6gPkyu;|Vy#Kg0^ zh3O0y+3ylqwL{u+$Sa78vr0~JlQkCNh$TjA*>7x)Bk~_RuN2r%O)S@W#8}ZIe`BkP zeZVFM0?4xZ&yG4+Qe3=%;SbH2>w4e%ThW{E(zt$dpuuZXN&pv3pbcRudrH*dE z<2n57i1R+@S!WFYe_!Ooh*6I>cJV@DYZot2a{z3N{+JEo8(LJ*J*z*o^aqs8v4Uqf zg3Q@EJY4l^0C$xBQEk!j1s_y*kPLg(8=jP(IeSodXD<1p{*{xhOa2v3y^wA?5#H@w zyxe zKR!8%Bhd7>Fg|%@G<-8jFYs2rz(-zAzF~Dx?uv6|B*zDNaU}O>6!b+Jg~hM{#910H%oyBvTXK9&r2_9?1 z!69VTW{;}waBv&1xYWiil{>*c%J}K7?F6^h+%^7OLIFItv01vaSjH*KOP8SeX-;r3 zC19FjkIp0>^Xjp{O31HNcyj#8l^iNhvx>4ZN;))a2~!!sHwfnR8-L znjkRBMR1KFVj{+>n7BpB6Mbd_GB!0nJF_sgP?((e9-5kToaATLO{7S)iOBPAs$9wK zbhdYbgG;bq>+3adedncC(Zw?|(w1<%w7C=PhmGwbhKx$B!C_r;+wM-~1mqPLxD5vf z$x?6|FrIh59oHozWjC3>J1?Qx$`lHrz4BhZEGWNiduO3*Bh(SmkiOtHrW<%0apmxKs&>30G=kDF=0YgAm0B^Y13S_>TxxG! zCiy1N{*si-@rqa*sTQ>WiTfeLwEPa&9LDAaT}(-#fnvvl1s+g*C^xyVk|=o%8h~1x z$4s$Jk{mBin+UWhKtRRVx{t^MobMH9gX{Q!!@I~;8#c|W!xFwmMO=|Tc$lv$s+~1v zCSMy7jmneRRJ_%2DL|p&)@p2}Lfr}=uJd?kvwdnc#5V*I>zX1ULSq6o>=aAn1eC}w zPirOAp<};JrGrJvPyw;)$YsTI!sC+&0Eiu|Msi(X(W%wKV2m)$5{=flb547Ym07Li zHcJ2@TtBKM8B3}lV3(iy@oqhJAtjtQpNHr`59Fgod*DO?tnF5tE@E16Sr3+RbcL}~ z>L7L$$Jf>BlJ}*yV~opp$N=Goa*bsIq}J{|$5B-*jy)ZoIT0iQ){edD4z`W_8AclUCwA(KI4 zgW&SMaAbC^VKh_*Z)*@t9!#m~G=NiAvB9NO!OaEX!o@GCOqk97iOdB4m|Z`s{(^2L zY``aqte^CNEL`7Z+qY zn3e~w!Yu{x_`-xoSO}$EabcR^4|%G|LV=I3?ZZco^K8py#G4{=r0C`3xe_K+h)^M; zbhXY5t2zR7Tgx4UoRE=I@>f>Sc#W=$Dthg84?@^-H<6L21e*tU5h6+jRvf=ss^tuS z;OX-t%S#5upT>YX@&L0uKmw2Z^p)dd+~p>fISESke`{Z@pRZ9%J+GLH^SPYst0 z8tWSqP4B+5=g4ZXrYLNK<#!}E?`2FP6i@i1xY*u7;}>(VN_&tRQXb`d7d);JIqh*-Gw({Qnr@F2&C9OL<)LbtrzfD~`| zB+}TlVseU1f;DU?394n>s-#He&O>bZdk6i)Ty{L@BAN80!(LRRS9CBl$$rYnB#Yy6 z)snZhxgEWA(!9zEnPS_nr3fA29ZsIxF0&)LBUiV^@<>ptG8ISHd7Inns@c~yjB=Qm zwfA<6QPf?c&n~Ic9l!V^a@D7XPMdz7gr*g1e=KyFVT1Dv*aIz6oTIwMK;2rmf)LESr>6sZ4O@9qXHUKmIEg3Yg6MlEJ)D7K&Y+dePvxFrw8ozUWF(oh?)4 zBvzdI`TVqj+POKBSPPUGhM> zcC;xlZ4FE)Pepi93%NCXI%&_GUWl}pteDBC^VxfE6?v8gFlJk4v2U(X=Qp<<%8mX)Y|?lH+DYTz_c zYsQ_WXCO&DW+k2;jFsj|ki+A9d|44@-83$>kFL`Bs-mQvrwJD%Z%2fon{b`k7W0lF zb9imi^IIeB^%kx#R3+gAFig0vyt1^BUoM^FO-FX3SAtp+rrSh)%5PvG-k^KhX( zGls?B#NJCZD?dmhbG-83^)*0bmVQwUkb=cjNgvGhup>uI-2%;GzR;h}Y^z%XvTs-t zgo*!;ff|_@fG>YSamzy>Y7Vsx!!;0g!Xq~kBi9C!wbNhPm!dl^-L|v_-j;-`(icH; zIq>SwMtD_IPEIe935n=k(Bz2!JmH?LfxYv80(hXP47{H>WiV*P4&eKn!|Tnq|e@}@94q3VqqizW~%Qj zQ+?SHSi%`(-Li$$y=7QK+sd8~{4KU8P;{5CQCI_ca>#2bQ^V`aXZXs~7ME10JU}CL zyfW0+fRQctCN*FRazpzpz7O$;krk%SIH{e9y;h4(df%dX?-XCHIv1~A*uETdVfNLl zVM9EtwF@1g|em>@HxQ+v1u)d`Fnw(pKeUe*0>}Sle~Bmnym}(Q{O~&>u~*bToz2Pgjt% z?|Ji(HhJO!y6=M#w6zJc_Ub|}rrTG$TgeGbEuhe6jO zmVxO#ooawsCGiJLeq=0?$dRH%B2Ti*5W_s|m8%81iIH86hu;rfuDWZ*l56PsE{v({3 zhMaLt&9F!CvnZ3^d8c(?<>yP>nWEhXq$e>c(XQX9L#HO#G6Dl0?qVZ|x8-yiPP^d@ zAtMa3v{Akm(wr!aSGA!b#f_&k=tQGNCe4AGpNI^X5{Sg4BcF~`>pjW~YLrEv7-bdI zb=ahy;QHumli7YGYcML`MzQi?rj=*{hM&x+w`0LAU<`=~d$J5gZ$+)q)XLa@_@Ew+8mX}K4!?6`G%V)FWKghh}eVms>zv9vC_N$q9n^UYS zy4GKr-QRN9^a2e|?wpbq)x8uT#kp*>k7VBcAzeTP0Om`R0{3PPg5mZ~n^0ynb1)7H zK5zsXv^ORV9?rb;oKksQBJz#E zRlDLm4Uc`N>0KEvx7&@@!xIzh>+9nkzm9Xv6Rmcq@&Eur+XgZDM7!HSzd| z&yStI@YHmEW@e&Xn4Ews19uR#=*(S@S}mUu#Ui_jCVW@J&zyg3L>w9L&^Ern{f`5}NEO0%0a&J*l}>G$MzL?v(46k^j z>LM@JCT@kg3^`kZ*H}f7a}V-moHE+F3}sp_7ProAZ(rSNY;SL0X8yVS_EwGW@?74E zZ}x0%kLk_XqTa;LoM$s|RaIynB+N92BkVis-ky&GB& zP6LXD=Qn2i$Vn#S+`D|0uvUJHWaoJ0-$q0zd96fDeC{7u;4+!`U-!YpD4N8?%yH|V zoCJT@=rje}y4+my$K+Pak}tTma8hS~99L^!ra#8UN*d;uHJ1uk3y7-w|J%1THUG>)B~zLn$uTyL^YX^F`mj`>=+M^lP3$m|%=pg6l8^lv zb!=nMW+b5)o9w+KBQ^KpJ^Ms3j7`lg-VM<^g)E4EjVv{GXcSBWYm+WO2~9xm=ft_s z238?tJd!o24!PwFZz`WY*=Y=9hN-f!PF#3aWLeXpOkHGE+vG=DwH@ow#>}Z!&GQ{$ ztBCo0j_YK%R08kfI4Z7CYf2MS9K1tizF7wE>Lw+e1bR473ugD$5nhEuuF~EGIymhV zTB@G1V5O1z-T}#@8HWQm1sS*-oS~oKvl(TvGHQq6sBW8ZstQrJeKNW7zDVE6vl$&y z`S_!f{RB{KTWglht$p){M$6r{&^HOS^R%~V?5TlbytcKU5&0j7*7q>^_*5}A8 z;Z^8&`pB!peP3HZ%Q0uH{7YF)5}*5@qzJ0P)h3DiMK}##Q5k-Iv1$|wBvo07N9DIk z9Fe`JFR3KQ*%yAt0<+1GKmRpv=j5f8^M=4h4m)Gt|Ckl~zI^NcB&SDEj2H(`Bdn^f z$HzPT4NE?XET(V;^VC>wu~n~94*pgdcQah0J76i&5qos-j8IM=%F*TGc32+ri0S6l z=t@$t@|RqA{-ST;`N~7d6uZxeC}Y+Z?IXfAWbVVf<{D2^R6-o2DGhb7?}+L3NVXzK z9emY>k*N;em&)^(XP!Ui^J6v?dH&cB<@+1FAxRJ>2zLY9OpN_o&LYFi+zF?t=$rxYk-*BOglgnF%xszfq1jms2 zt|IpRq>U5XyEIR-@U^lsaNd~6Np@laG@X?I-S>kQX3-3xFBWk(se4WM&5?fHzm8Qg zMQpI(V9jaX%}XA-#-U4VSnIZ`{%X0+?j=IHMjuL+3#NA8iwLCPlNe1-;R2cGMaZ%I z3c~$essj{?#BX&d`6JI(CvvldNtGxpqs7ABI4p7zeIh+}+PbHoT9Du~5ee3!j%^<4 zaFfvX$>f1&BJCzm8RV=kd|nQ=0E#Uu}?OHEN(^7zKE8CrEn&T!Ph{` zroF!L(blhfz=;|_?9tYq@={1oa5&2H*&NztW`Fi+IKeI(u#lnml3U|XwJI}b zuJ1GsNpsbIm4^T{*S7d^EtZ-gc1Ag#`OY@MJh3gJ0eUB2R8!9D3@SN#Y_9l@tmcYS zUzX}a4UaZgoQo&hdO}OKrCo#VAULzuChZ(5e^!&MZMrEMmzyC=F!QcTO`VWT>r)Y7 z(e$ekb^e4snHFA~^!lMld%gAQxnl>!a!dfj^r|hkJkVZ3NJM$5CzY<~<8t4h%nW2) zx{-EhB-MgGDp^qOtrs-P{+O8!(Obtl)syK+*;K9d+DIA}8aEZ(Xc56+o1!p45 zSI%ym!R|W(M=0QfElMfOG|X_oTrWJ;a?wY``gO#FxX5$vKvyEpePzYTQ8qESX0WrQ z=U7*~GQV5Im3jfs9;QWi^Hv}FVUJWca(R=lZ!mTkL(VEE>4@*;;R=vu_m zZ*Rq<4WCsp{q4s)O_E-^uNT^2h1O&UYGgi0tg>;LQy8^I*F&I~%)K5@S3OLcgT zpwQNFsaL6vvk>B(0|^vA&M<-`aj(oB$V##BLZ;UN!k9we?|n**4O z=z?mK3612M+Ko@hQi(+}S@w}!nar0X%ztCPq*T{oAUd6)X}sFNXzJ?K?;A{iIl}bH zatn<0Oxn!bUZlO=GtI|8EBQkJ!)BU2G%k`SvHwrQ6bHDsF4Q_u3}{H{;(Ko-L>n-P zm)J?YLCS&iq2(0UGfn)FtTgdwep5=C5(G^Xzw>c;zhkQumJY3m)5B|?Jg}3Zq?Fvs zcacgsUirIyX$G2JKYPMsTezJ8hpwPYW17L)I(Zpxv=&q&_rurS#q` ziSxqbPjrTe91FU+&?V1Wzgey}Rr2k{8}z%tk723IU_G;hG5SgNW+M9OtVHy44_a7W z6VWH7?=rj`(mdEf93oQSkQt=qzKvQvYOQ`}tWBkW0?S9p@;{EqvSu|LiE^y4_TwK+ z8vgT0o4qZEFPxRc5&$vF;aHe=BSHZUt8OKdQQMV4^!d!tx>-x`R2kaCl&Zixy~|1? zno_D7hdpxMCs&U+^5a}yg*$T+VZ=DGsV*Y`XRX13#2rvD*2ATpc3;ds-i}6e@diZx z4-ukh{Bt-w&JcDixF`DB*ISIg z@@dI90vIO7Z?cWK>MgK;x1nTlVv3%G(uLpPXq=t2-fUlOeRUyU+O_vwX5{1Qz6bAX z2&OA=g5p#elnGKN5q>4&+?+IE{Le|;K$LvK*>lWsks&Ofg)a5vvWtdkGJ-B8xkekcy_q z66NCJGvx|?l>aM*b{R>W_$175Y1VPH4e-zr zdW7{YtGX?UxyI$)N`RGLjIfwypdJvcEEre4Fxm7wB5m}Z&Hh5oW&!{Sme36Y>eRWJSBbcaHF_X1^I5XCI0CJT= zW_7ZptdUHXL!lHgB_N1kzBEz(>C8~JDSZpW+#L18sxs0pITqX*(_R{>EFY7pA|o`D zBR0R387-z+=7@e(tQ?tOcl9kq=snA7%F#B<28*QcX~t@~*ZAy!NwVrMBk*q&lO{_9 z$P+%JRCu>I)>!^(O>7KelG4n0x#AMs`O0-5=~?-IXk?F9{0AD#Tsta~DW^j)`~+DZ zV2WiojEoS`S?sDO9wtGX(}F?LSCAN04&TKF{1E-q*G=+A|JVY(IpE)F40tqr8D2Zf zrSj%!7rGTmOKO&KL&EUBs6fcq3pkSajGgWQ?bO#>23}*fpAea2X+q?lQ?G^ev2>c&X?K;aQ!tc(aqUBAT<&_DaoK-qV)vo9K8ar*Zsg`XY z%o=jd9?QOM%+>FYNUgS1JSd4K1Ydh&a>j=ut@K_jp2NK&ER0ftz!r;p+4yK!hc9fP zT9!^RbdCj&Wir+pL>V>MH%6M>eo%3q9Ed=BX=%H+iK-2W)uN~%x-5R*9K~Ql<`8Dt zXnHIO5E4|c;E{(*A`*X<0xH|(X1%joMp)b$6HOWv%BAWtka$Z+vHYu#W0zaTywiPMBr7cBl_Mod1f#p z)@XO4in`r%)nZ>i+$o@RG7_F~hu@V)x*LgQV4hM-y~9FVQ{Ic_-2jiiT7_ zwQLAmG@*b>(}ZWT-~tWmY})G^AI;P(z6vZUMLKvjdb3bNsmyNoNodSS%7eUJV8$__ z@D~3ic1cBEBxWS9_|SSzwO+sGKp?S8L1N1e8oj``G;u64RHvjW*FES5$Z5{-(uw?O znDN4^Si;q?y!XXCUgA1c+~ndmb8dCjD>hwaR(B}R2X#P~cfw*@tvCI4d5xWigWXOB zJBxy2u%nS{xnX|v(s}nXlsU9@Y?s6-3#NyV1er0geLT~Re|?VH9v zrUUvEIwPUxYNYhSgPl33T{>HbQ`SvRKor$Cbg8P;k&y@}?v&w4LqX@|HuNqW1k1=2 zx+*9k>#99lwUR%R#c1l1oJo7&-Fx1|*-z6tvCSS!na46=EDfQI8tfY*O*fqyoXs_a zvS@7(Y~C}7#A$d8*3KXpoXJA$HB7Q;ZxDe}r3y1My734MF zdi;ArvYMZcu%WgRH+VZ4AFDX@sA7)Q&qv_sEe_2u$w3iFFmbqLRL=}YPheL+xmPcaYz*S_nS2>*I!2**@h4B2DBt;9T<9_7!;gfYfHnb!nO`l(`)<4>|9hvj5g zlp?+LnZ{i)eFXFN*HHRF=mq&%_hy`z!$gUO{x{} z?FP9tUW|MIQ|`SxvL;p1Y8<6Wdluwyq`ltCrK6OK0EQ`-y^KT}@ZpO-fKGb_`#UoO z`a_wsokoCHpz&m4J{{#^^V#6(o-8jESGJ9ss%<*6fsArr#i6Ekd67#YB{2qSB%(UK z`juxZy#{X7q}St_(aW|}65=c@f!g&iq0DTtyUKDjE7v%C`uHcw z4i`IWn@Z`fuqv8~{qIhBD-;?+3rN4U+`+LQT2qp1S^B$C0NLeT1Uic_FO^nadb~w~ zYPf!@9km)|*OoGubq%M0$hmI6C!K|Q@hhSDtqk1dylhgHKwGv@mifoA@}+q^JPzcC zi9~1#{B5rjBsKN^x&aAlZatzHY|DE~ne#LnSr3_H^57L|-h!>tINdgtwoLhnZYl*Y zo$oZ5<3(7H^0JNfdU#213um~on$Hs?hLk=z^LBvlRXV~23GGl14y3{Gm zRNXMG;nQdgo7SQ?EyLLTDIn6AF;#Jp?qew!C00hwnEOK#h{;Yq%Ry)Eg-=I%>1|bh zL=BMWzL!-g%8BVpTc2~V)Nyk8Y=ZeuG6QoTKS!<&h0c#_3j(>kFie#*Z@c$U~JR*4{wv2U3@&zAj(RIf?&u z?Haj_Ogs|1j+C}_eHoFG5@qS*yg|UtH8>_P%NXQNLqN&wOqcu#7&OVwOzrP|@vo(8 zQ8B${jNZv_5!LKXF;!T;NEj=BLjrWX@?ZOseF_Br`2Dwu=ynB?xhH(qkm~$D|v^OdI1Gd1(@TU1{b9+SP&3P@lmPBO(f7`K)2L3Zo zS_rKT{Hbs=OIR~()xee(h*w4o9baT2K!;3DX%mUk6ssrKR#`M9sZe=4C;-+vPXF|E za_+OwTQD|R>qC;YfKgOAU(xT_Ov6f73sUg~uRzO^B#bdevSN(SeV>I)8e=>NlbX^^ zG`)2MLh*>KxK*r1KXrMDPvzZwh!^@E;xGK11$JYI7vG%a6mH{^S8dL=C_7~?#31WA z$4D)AgG8DQW!Nzwkz`UPbbl)=p?mocER3s!Zt-p4+6&CnB2_IeOv7FbzpYIAh3G18 zrliF&C-$)B(gD9tD@XT{lfx&x{4Bg)ty+hR2?2#j6@zd^I#YxLNy~nVLJ(=!uiSp9mtp@y()uE>qUKk0#S~g&eXIp2!|(IW zw!FwKHb>HmTeKJZhzWFC%X;y(0jwNKHeG!*tLf^N8Ova5)75@CUB$Ot$cckBL(N&m zWm7c@Ef43gAwl{yqo1RApi5)<(fk#5(GePTuh)HKy7bU0B{hpoa_xHEF*VS0UXJWo z1`D5U!ih`49L_~F38I+OR?cJ7?F%1rJZ6bp$(jQ=&u1ppxY&d0N)Sy|?4mNH^!uap zj&xK8bGszASpDz1zR8jQLj=O!j^(45CF=?Fm}7Z?{PVlw0*%ZAG-4ySB_>nAES=oP zf_D$Vg&OPwHDiNbF^dt2vSIi8;jT}Yn zrqvERM8*%&ojB1zFoxSYaYDt*y?4MimF5WsZVH<4Zz2L-{vB~xG1$j}7qg&vlw+X^ z9IqUBb3Y!+--h$M@S~FU0tPm>*@AmS0Wp#ERuWq4){*^Ti~>Ib!LMJ(>OVYXNIr); zQnBhWCgL1q3xO%fyLaNm2b#!Pj$lvcY0t&^&GQKQ!0E&jCy-kYU-{`v)xe1pbVyMw z^2xa3Ir%15kP4-Pn32%*GNL{gO^fP1g8;1Y%wX++Vcr=TX^mly*~uapsM}}7P~A$P z+Y^y)lZ_{0uvx`m0%pcwx6&YK6yeXwm>r1@slX}zL17dBhE5*I3Xk$W*%Mpu2pols z*!(V*GVe;onHsrLRseb6U0FZuLL8K1~NNie4FzL6d-UUM{ z6jrwzON%!KgE7`?JQ$`}HzC%USaYX-Raz8E7&U9|kK)I{=2MC~?_@IeILAxV6E zv(auS_%KVwg|$AE1HS@G#QN!un1mx@FGkdK8fdre`-_c{d)O-^3EQ4(MEI1&TS?gE5v$d-zmlQPkQ<+ijE!|{HgSdMc8wQD#(L6IxW+=o`PJ%33 zwP4rV(9uLr=~lzg2?E3E61Q>?8(6QFT0AIhV|!_mY#Q?w^SYR?3exfsmPV`=zeCa* zgZDZEN)9Mf6u$*_&t&2&zT@`zc9S;o-Nj}xH{68amnsUqZT%+Uto$bu$>Wtj>Ps>y zwypWN1tybW$_B%LBmAou;;uMm@cPoU6~79jzRfl!6VTr_tUnp+5xjaGN&>Qh{)?;z z`o%R1QPn^%-ix(H7|Mq9O4{Lw7CQn%zV{j|4iU{n4v?Y3Wu#1oVNw|4!3J)X4!}&r zLmG6W!$p2lq-^zSUK5*7rje4fdo$u^+KqG)Y(Il`?O9%B7q(RhSA*b^fgq7hD(eNI z#7W_Octifl$A3Mbw0D9J{oQoYQm-AZu^={hJ#WvCBLdSCK6{dLn5{~OKL|@B>ur9%%#i-BaZ6G_5<-I8T30Mno&R=doLSiQ&}w=zp~$D9p1!e z8p*^0fZO=qW9S(eyC9jho9w)+(W4|Xn$T073S*&VIL(Gkx*B5 z$V@t3`FLN}61h!2`?3W>!)>y-88v%i@l7_fr&mBzZ__3kwnX+)bdu>Yg{EmjO#)KZ zE!ARn1ha1PrcsclUjGosx=sgiOg#bOp5VqRG2vumX&3~`u(8DxFU6WsAV-@?d>+hD zu?z=;Bg#Z+2xF}n(OhZ!>uRgy317P@azhAd@nade2b>#1h-?QI-n$`$SdZe=d7ldg zn-I#0miZ+9`>f1^&mBCRw2!j*czDv|Y-g>psX{+boTxR%VJIQ|52yCn#q#i{b29>F zr|;q}9J$Nyu*guBZi=q7M=07*eqtEg79&We6P^-8BqClSPyk7-P?hCV#$kRmO2Jxn z`0#O61d=$jR*)HMx$UfZ@a>C(iZ~qL;V@$rH!V=0L5uDP#*0pEndP?Xv^aP<2KaSk z{9nEX*yZ>2vcHVf&g{q~B?AiM3MfFbgslQxwSZ{Jp?v6{ukq6$IAuv!-XP#XZ7Bc) zZakR>NS@nPZuq+h)CfH_7oGJp46C1r7*<*k;yykY!d?@Pl4X!Bf{hlXkNv4gGs*sR z3M%}$Z9 ziQ%vJMNEq0=tj&#x+G{Yo9~l>-U9&=0JGmFuz#2tuud;2)ez`i)`nE#IBPo_kI+c( zUL9y~4%*THSVY5GV;rsG(+@(Bx%LtRt-X=M9cV8B*P*wX8*#fA`H4O?rbgestJ(;v z1oWWP#gY)}jU)sUQl~HUlY}(3Bcn!20aduF&zy~G;;a5?Rat6=p^ym~J`Cv7st%F3&^SOFh%o0dXYI?Wm#Lgv?a(|un^bWrDiX#UjV zVNgbz@7#_>7ENtWjq+cLj2#gZRxAmIM{xAZK}aA>cFG=v6jCtZDGrC-hAWo~%#v3& zJek_e@@{!;`rTs5ZzIX*DVL|$Ugir5w%lMqc=s|k z^p$~kfUnesM}%3UUPx6EsCQ`W2m8W{7IfW)l5RNaDYhDp>B`&iO&o5`lzMxYg)H{c+|P1CI$9EtcuXPe;v#2wbPMy`e>L4I;Fw}F2_ z?IzZH^|kR@y}9OAhg<#&-pSfXu7Lwm?O`PtBe_i!pMGfS#KuD@D{YR91j?R1Kjdr`w=w+iyx zhOa6Q;OCW{$_f5vxH5_#!M;`-w+gTARK~-nQoYmOsZ7$-)v?L($*Jju>A7kAZ*n?6 zH8WetO^wel%oYlTxyhNC!p!_aK6mMolONB|OfO8$E@10vYJ6&8dMZCZHFYKDT+$xF zrQ-`Tv(xjlGYeO)5qhR9#7w1S-{_ywDRS>T%5{p z+s1YF_kunS{;$z0I0QTzo%Ygly;E!NJascs{ywx~67L~Nya@3hN6{GZ-%j<9R|Lh& zC&b8^`l0env~qB88F$SuwKp5yPH^jTz2xl>Jby`pdAuSNhH6h+I^A*$PsNDSkJp(h z5chD<>{4VYFOW`HuA<07iHA-g?NyHLL2NpdGEtm;47Ucv@ke#kQY^Y!;||$it#Z9n zMT*^OtDb9hifY}WY@rrQ-N;r<{3v8J=aJfK;TN;AbepZb#K_0)q@!xp>#z znz7Bpl3hwtu_s`OB$l);tAf;6I);duYg(0*D;xo+Y1(GO!j#_(T0?Cu``3tC(w(^g zI>t!RCogETdp7NXiO))Pc9y3-vRERPGLL7&mKsJ`wb(aKLK)y_%VU{HiH1~04Gw}6 zPL5!awKxb?Kk_xBLNvmC4FPQ%8?~^N`v(QFF3{eX*49=;znJs2KK0Y3SG1fx z`md#pBoq&G5zW{p$le-Qf)nhr0ShTiyCI4MMI&lR2r~hO-Y(V^i{A-DB4PD#f0!Lf zpLaATe%3!Ds};Uzk*k#yYm@tRLuP|bzDVuqvHTEO1B><;yc^CKt*}Dh0$V|CYOgj& za>5!|3KwID85!Pq-^R#9O?@Kd&rJh6*ipDpqqNU9vDotnT>|9S#A0fFispCu|A3t;qVZ z<{<&n^0@POr-`B;;fn z73o?QUPrA{PYsTfX7Faw8qBX>m7>^vjcO_Lcs9h<@X4yhzHt&N2y4{$4LsITO=i%R z1rt-^Y{qfn_#H{n!oi2gS;8~NA^>XHD^aFTmWc)X=j1_W@xsiFw&r4S%bZEe^)-0g z!RRaGk3pm^HrwMlWZ&AgX>FJj3IX4*M7dGzn(SrJi8WF9q{uBkAFupXqzzNWMQO%z zaw?h$0NAQ%h{Z%eet!!2F;WHd!S+b~swcvOgrH)!UL(|gT2DsSyNh^5yNc$9 z;K=xeaFw?>l3Q$)P^eKFKjXID$D8h&H;lJ@$%}8avX%burO@;^>EH_vbTpg8qG-I7 z*kD||K^z-#Bx|!p%!eRR11w_BN`Vv>gej7Viht??A%q+~$<>#@B>EQQ5VR{-1n1%4u-1Nfy{N%y{)tW8L z&Q4BEO^2WIlhgClbMyFw77BCuIc$E0pJoelv-$iC8qAMl^^~7lD9q6(2km2%6deE# z?1$o8VQyYDSePwL78d5FsTTHE@pWz{Y;bCJc4{&|H&1=#C#Uj-g}nGQJwG=$KQ#-y z^5b)bsRDqVXMGjM7pTh2+?3{ z9ixw;POd<|8ENoc5GFUpp#xPIoSY^(pfd^5fgq?Q$dQ{t=bRRNXP784pPQw!M20C2 z4Tvy1nVX}xi~_7*j(?8z!H{PJBh>Sgxdq1w@diZ%)?G2is&tb5$DqOf^};MLCdk-t zM}B|pHl;No@uZMMge+;rDkSSwS+SCx;BC;f!)=l^Yz*su_mj;}kDG$>9&guHv z8qVeMN!c2zW1FvYIV_?>qCFlS5JX_j=E%djgOp-8kJ%GKI73GqL9XGkVTjJmiQ$D4 z8)F5;{T4>xg66OAZZmv0!YdbIm8srCc~);?dSM*$hst?sZWFJ=YE!J5)+Kc|N0la~ z!zu&<>u-v+6;_-^MG|cVEY+N*BIt;fXU(8^(_?|)9L0xNHsylD|D6i|oz8LMEF(mv z?JX^`c%xCL<`8YLKC5~NtR8Qb-G(>fJmTc>_F(7>Z{l!f6;aoC6S~3q8+g)l2hRdZ zRu`YE_)Vcv!!@D`{uV|i3L^&(-cM&Di(Zojf>%8px}VB)%N?Xh1)s`509L)>{3ta^Fg-8&@X>>hm*@ensr)rno<30NgNv6Q9)13yYm`G` zc;f_Wj9nW+MXXk@Q8yvs7D1f*ooC!@6wQH?spZwCUm|Eru(KL<;KWm3(|VS|Be+;T zh(L{y$(W@$og-ps;bSd?mytymhLLq@KEPZXo*Z@Zqt032DRvclm3sm0@;)`LMz~mx zXcrw(gs9OJmFr?H-tTnXP$wN^EUCoKL%5;}TO=!Vj9Haz_lE7Kga?D8M(1_S<`j*E zWu(b@L*T){B)T^QUJ39Gfqw%EJlPqa{Qdk*>^}-Nq{zeLl~2Rk7+^j1_oT71i}h6T zyQJOnJz+c9uc!XEgE!s8Rt?(Pfj!fog8YRwNBG~g_~F3j5C4O09{z`QkaHPp4>G#( z7k{KZzWZMe8S%2gxtS>}Na%6CFgpnl@e8|GvjB{K(ONM4I)k+ry_>~4iNBkjpP44G zvEpH0!h)iig{c`@c~R5(g&E%T{5+Ot?A^lj>>S5wuCRb5B8p_xZWZP@hOG?tsJ87z{*U$kxHLJMm43#4WZ(x?jv6^S+4D5MlLdC|=5 z!u%X>aIP?4AZ)QP)EeX?h4$w-iRl-o4MD?^qF)>u`pTQ3U$gmX0uF0R)gTu1Ga=jL zr)Mx+RB)lNz<*(l8?p#CMzPRmUuWh7gbUO2Y*0W_iSpb`emW#ER`%hDVqF~mnqQdb zth6vQ8B%U;9>c;Km@mxp2J?L2*vpx@S$G7f{aLs(IN7l?yTGBrT6Q|*hN;OA6U12O zLl&RIa+y%hv-K`7h`JQ0O$}lI_&aO>&+~V41t7q_&dp90I2pj;b775za8N>q;>{G` z#o$#YF_8RMelkRPZU&h0cc2GHes*DMKI{aZ2qIrDtj8Nf4NkAQsp**zt|^QkZ)P5Z zV80M#9`YfEF_h}LnMtB18sr`D1|eo4gpjhZX}Ed9L4j2Brr`kz#c^)BAi9Hdg*Yas z`4{M!|3_W%|7gs@UuZ1+g$}}BAVK&Ga0ykIfff`>P%*ONK*c)Rn9ISMl~e|U5qX6A z1NVfIFfUnH<2}hVb858JFmd_{#ixmqlmU%-71t?tQRB=9NT@s|i}O;5H11bytN2YZ z3Ljyi?36HTQYgho@%@}+SjA}}i@=G+nNev2DaN||DZ!B(uUM7ym{4$YT%ABJLp?we z$_1Eyamm9k_`t+8!+D+FaV8b2Z60DDipbm~sb$u{T&UF`;xnQ2%}h@Up@Qp;3sru4 z4!_99M%CtJ0}J^nA(e%~ELTZ$Gnn%@A?By2ct`Vj@FK4}gP&+BRA;2_LrowXn8jog z>I67NLp?XWFdM==HyO(4EQO&^Qz5t-3IYPmm7;-}{AAeFTwzY2KO;mt&*t2mm=iS` z2oz>D2vh^JA%b%SrecWf)0le*st_%BXax}k4WK(E3K9aXfjOGw*a%L;B`ehyWROOI z27nV0zd{;NVU7a*;tsJ*!<)&n z1F)@HmLh=smk{j0=5Q@te4F6jz5a7YlZ<=)y7Dnx#`5IQt=XbLGhz zws}k8GA)0#CcduI&M^5d;5daRDL0HAVq9X1*4ymxAL~h^45Jprg<}+eQ{Al9*L*|* zk;lhdZ6baM7n^A(Mcz9gR@%1kz)fMn^lNg((>?aaQU+%PvY@SoOja%SjgwgEQS8m2 zL#^R6cw071WQ=|#>AJ8#jSQ+O;ZdJE1?4w?Qm$RaXtRC#Tfn)}r616f1*+aV?ppK8 zrLRS0{3g5PhFpLYHx~xUNiipfEQAgrKT>eleMmE+Vwm`ouaRElmR*VqLQGK+o?+w! zq*GrIUc(P5H?>1?=SS&S1PHbPs++k|v(8Sv>D^He@)BzV+G!iR;!+=qU&7hO2%swR z?y*cz)CMD?1~H!d-e$(|KoVj(tSn{DWI=8Xp={dg8y`*nC;A+F4fwvR4g%-BA{4X> z&nb!RRnh`QZ)v`+l!aBAJDB1~pu}xjky$$g9y4cn$V-Iw(T)bWZRLbY70oNO#2js; z8;V>AUVP=d*9QL8=nbPZ8I@5v69h({o3vL(FGSkwJw<#(jk*AaO%YwLj230mhwf<| zD7G>ZuLYYaZnarK`lhY?_Ev3MPf4}7ef5I*e3IX1Q!H*hX5QQpFM37Yl=m7!Af=jl zy3b(%lRVwX=2~BGx{aN0$1k>Jx^q??iJ)Wd--zHfO#LtZdpY$hvTJkm)eCjZpcH8i zbGz!{>~t9Hxam~u^=l4hYd2y+N2D`;!JWuhSkMyQ7W(f;6T{uqhFBFNVxZ;8>Ul$% zLPA6Y_fy3j&1(@jdW+-7_Z|^J)&hx6vA6Els;SX7h3r#?kj1YnWtXx~QYt{Yfih2d z1Mwp--sIUEBO74sT6bL%DepC_8+{pXM~UI<8b)(yzac9@{miHwaW!SFVpREVV$|c6kM+f`?Ea4P7U)gJ`wf1JPb!U! zV1>)6HT~tHx9Zonirn(p4jK4r;UUhzS1Ge5{hZw9j2ynF2aUaX3HpY&^uQ83(&OYd zZwv3QQhrW@`A2ht-fvj^Bzv%#=?Ai6_|LsyVHu6#Zx8b*wwy|*)fQBayp^z3{t0K@ zAM2ZSzwm<=I1JWh3GfZI8M|>WekBYpq!H2c6X|&LtGx{}d z@RG#~L~o@41c^ysXIjwiN;GXU)fXdp^|oKX@@2_Af+$!{s#WdG zt_D_Myvy1qQZHwO)G>33j!dDlpV*gV9=SJ}Q3dwBiplPoLio>_p=n(13$KjsJC1*| zpxS%=qOpzD2!8Gfb`}~Zq>-wtE-qq6wne^v#Va_STpYOGFt1)&+>WnO^=sG}Y3X|O z6X`q9hPj_wpv&R6M?A7hE_$vp71V+q`l({F#+xH>^cL%{DzO$wFfWnqbf+kg-bJLf zu0~%mO)!n{YIfZYg0IIg1&pc9!ZG;p@hIK@VenB$d) z@lTQyh;lC70f7Eojmli4N^k^u2dV!t>VK?ty`<4YeB-CH`L_tJs%e3%HXipN_I=f3 zq_4WIR(-5VTcRzxCrR}})T?~qE=uh9z!gfsNHl!r=!0yN`FME`QLyqwd`Gs(Vs{Qi z&dhoTsRnn-4^bO82ZxtC7_C}+DSUr6I3(TxWpC$fZ~!|)w58kH3GOV_mpg60+8SSJ z)@yCtZMPHbyXaOsUTfzQJKGpPgd8vqOvGf()?FC%T&7I)rO1<8uO>XVuQ#S{Z~z*m zc7lVQ_R84&7&g{-g14C;m%L`P-drj}qN-jKbS6-j+dIJ>VRBH_%3jvWPVn}~mq-(p z7r@oQVcu@M_S?)4k;=gl!l4C1l|iB^aM}rOui~so2dRBQabyJzZt-e6!Cg%c3Z%4D z@*3a|bmKEh@59dE3eW^a{L&JVK(6>3JLP@BVYkz+FR53-Ek2m)_7XB2EJ?-#Vc82P zF0Y9UA~bY&us5J?twqDHUNu@?r&M3sSR4hZ$E3WDVjy(zlr zns_^K%I$We_3*^R`uh5KV-vfewefm$b)w{5pFqjh#7=NG)u0xs#!}nEJ}U%gCpduj zWAB=r23Za_SC%8)&>Pjw+eA0*dL2^gd+in0O}V{RMHdI#ZV@6-YL|DOIuN`a47Ii- zWlMs;)Itsk4=(Xz3>sW3ha9a?M&MhbQ3;HEThm>aj}W{gt<757-B>F7tK};G4>rA( z^1kc@hgxecAnvR_-lPz(7#N*aMq+Zg4JcsL7Hf=F(nud?U0(c0b1)w+9a$y;;% z>XKV3k-lZ{?ZI6ZJ7+t^-~iVI5It}at)0Nv?WMNcTt&`T8uL5R5xAlag$UZ<99(N> z8@kGPTf>QQ_8P6N$cKHz{U!E(KfUC?hXXzlup~HaBEa6DoCwZVKDigEg+ESz{^J4s z`F;BH;$i&x2KuvnH~xG*{h5Z8SMv1d8T#`i{&aAQ75bqRzSQe*<)`qHw|JJdc9ykt zmbG%0wQ!czKHDkZOXbg@eEAUTz5EpK`uhkH*7Y9~B>x6ePvu((5(dPV=#c?ICI|zf z!Y0+=cAeoE4Q^ldx#N{so4+BYAj)xK>u-2D3<20~4Xz={!>s6=A}?iWoLS@32rG6_ z^va%&8tp^BH!1IOaCfw=@XJj>4l0YKXC~cz8v=c*U=J_?rmg*_DQz>}e*(0wyaduc zb#v=mcFvKSaGTHw;E5I`L? literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/.doctrees/chapter4.doctree b/doc/LectureNotes/_build/.doctrees/chapter4.doctree new file mode 100644 index 0000000000000000000000000000000000000000..701796f1f110e821a21debd43a5f70b5c9294b50 GIT binary patch literal 281174 zcmeFa4V+}xRUfFa=Ch=+WLX-^Zp(g*N}B4Ks+#VWWXqCT@=VW+zD6IJ8BmAmQTuh* z>#nM)FW0M@?$*%6Hs9e9Sf>dld?Yvv{NR@j!D|eGEGFI!uxyrqe`X0R1gs7Iu_V9( z2_Yu=W&h{g&-d!qSJ#s)%s4^Q_1=B=o^$TG=bm%!Ip=<8-*@i4Y0pjczv#BGRjxEP zi+-upXqJL-CpuVau6FA|qqFmOccwqQv%Qmx4z>M_u-R>|20PJBXt7$Z)JpB3v2$f7 zdNVceRBB;Set#egRy&nuLw{c0SKePfaAhY~J{0ZiR64c5dKJC(d~>}Lb}Fmhg2`n*!GCk56cVfl zR%?D3;vxRt*YNA&uY;v-%lugJSFg5%HEPq14s`r=dbLx&E!tbFG=iPy?l>0hSG0KQ z>~u?&pnWo^H}K!O-)Z=rlWSYy$#Z}6V&USYXCIh-=%JJ4xtWt|&2~@=LIN$k zMm?js?D%2;xauoGd-maCz4hP|*V>)U`D>M6qu_^yjbNqauLsj=f)0UF0SFJqP|Q-^ z7u}u&Ry052c%kRe@eVY3u)L>ybNTr5<>}`!ujO}QGRyA*Yk2s-qxiop{_lQF=6=A@ zZC7@p>t8~u>~zby@?3fPxx0>)kC#sX{33xr2ypr$Z+~&C_@aD^Sb1jY+^!zD*8(q? zBkp&a#^o0Q(?CKJJhuU!yA=k{f+-7X=*6FZ(@i%$-SB83_-(JP*Q*EF;MM&~!>csF z)Ir$sHvG_A#frqQlGnsHawT}3a*$bV25W1T)e7E+#M8dF)@`uR@N3>$rQ_v7e+y4r z&DvI@S+DrDY4q$h+ohoGb(-EvknvY)0sYxR|CQAabqj#PRuk(`bh+AWbSjN*vm1KX z{C0&u03czr9(Y}}3L2S`-|>syC45SQqBrWr^<8#rxsvrNp||3P0w3xoQ)!ffR)GIB zI(m%I%Py7u&h^c0d{Xqz_s|MZc55B;u}tCQxDN7kXHpUfA&)CBF^g zwM&(T-`>K?_d(c;AM8TZfoyH>LZ!4GWUOh(fk>cuor#L&->$5pCGe^=t_AJRwCFg% z-xv`6fIyu}5au&&zp)Nk;H|ZrbtVE+49nUB5cHrOtZXUjtU+MVE1v6XevL;%n+%Dr zdH?_v4VG4dj=!B%l&uFo=+E$%yrA7~wu|1=Yu*NEju!&P3{63+-7Ix6sbCHIiEXK} z-T;FMgz^psT(4XsOv^#T^D|+k-U4^DgVjn4QwvIAdch0p%%?o*ERk~MBhf+%{Jf)A zSUTNdl`h$W4`A-em_e|c_BJ>@`I<5#L}~Qq0J0wz5SG)=-ULx|Bs!DE!nQC;r@!6d zGCi_c<9qaLWBF=3I!G>4lZxzzT{e`{W4Hu{LoXZ<%=+7~{Aec>HyUts>^342SA zShSFFyzBwGKXY4bal|cZt)cq0=o#H8<9)4PZ41r{npoA%%| z5)Btp*ljE*Q|WG(n53}h*_fo!-R>|so662>;gV8^yTj%dO9QZAiaqY>*{sJ~cs&ZG zL9NKxv-q0)wTahcejLrjLTJ42dN3YowVs|_5wajoqqp7RaqlSFn4Z{Zq~=LTg1(la zGO+Z<^(-+>%&2L=s97MtKm{g=Po4|XKVe?f!@O5+Zjid1_n>o=Ww08w8&KUhDxETv z$<8XQ0H~e*RVbINxkA~jliFIUgsW|+!j@T(s}zGGG~G_H-fTnbo-TSX1yE|YSVIjq zeNtNU84OP|4?_dy26{U4{3$OCKG=oI%^EtiS>J1P$wH^sEkI0W6e-15*&jsqQznye0z*_OI0Zwt*llYUATf8hmH1@^tzuPgTh~XiHaz1>$ORmgqL| z60i_Tsl%}`vNX@LsVqnvr*t};1gA4;hzAR$bo!eFqjP;_)vF)+?$M={jZ@zq$B=&T z>$=uS(hj!T*qk?x=meT>;c20?FiwYkWbs58otjvXTiB$~*(9^`_+(p(g~+5mO@hVg z$#*sjlQepp1e;s*oM4kRI%Vv!qyORR<1yv;=usb||ML>|ym}L!Stx^b$aW0_IX0oS zHM`^mC(WTjPyAM^w&m62p_lEwh2GSG=a@8(&IUZ&O{k>gaM~!4E3g+S@SD3QsRQn zs!>hmCqm02{)8Q2x;`+1GU8}++Oq0*Nm>hyG$4%(`=AYwSj|} zFw+RSba2uLusqvW@f{g*oidJF=#(Vvw$KOcs1lZE(9$PfMeiIe1HItYK^WF4364qw z_Q5N{G@*r$XR9Pk)PWdvQ{pomSbfb#L7cs1U=V6^hQ6^JFXH$~9=(Xut!Ar}93>-d zTo|M^oZz4j9HHV=nC-$CR0d7z3@eEgeEuS{((K?g9|T#w%61ZnSr-T28heXMkY6|+ zg7kt>O^z2fRB;Y1j^=nwpSyw0TI*B_`qWO*LXIye0@F+!*g<~5!h*TA=Hc8N=ie}H zabiS=j~-5=%FR+Js8B;x00c2v@2GX&EXjE`l@({9lv0P2;50R0{acU@ zY~~t&x4ziwKS^-g@S~)MrL@+J%BM+h7faP4uq?C&G;$%;>u%L!a-4=bdR$0BBwA3W z(%mHZoSs}fTj-?G+eFw*zEjxPq|n`;s8IoT_q_w+;i{q% zKIL9}?jOM^PsWu{3)vtO-dv%ab((M}hwR_)B9yS?k>bPwPAZJiHt9&$VD(bihR`g! z^bLH_QoFE}#xtZeo|j=#8G2uXgB=0{;w) zlcc!Brf+Og%SLbK)OQj}`<(@6fW<7EAmwh(t z$pGYOvxFB?0NmpJRJy~AoTQ1df^S_*+0kZd00}K92R3t!ZWPnO0Zr9Y3Al$VqR%Lv zQs}TEdX(+Nfpcx)GfHb0S`O7zK@n3;_c2>p1Ih=kJ`Jf3$JkIU!o*X(7AYRAdLJcLQsRKkKMQ!n^c|*A9H$lySw2+e?+k5{3D#tX#ja{qpt35}Z4TLTGJg zGj+cNWn|6EHnKP^r+@;e5R{WZ4@Eq{s^6yQ125akdTRlWxMeW045WrLgiSE`hbnc7 zI<`R8XYwpxMJffMB06pa8#1n14V_sDYR$&Fk#+CrPXja6(V$kCm~TmNflD$u+T1@H zku5kz{lW#4S%e-3H4T_G#qPr!YF_1~`aKCc43~7zD@muw!;-G&FjBpZNuYy@wJZXe zXM;(wr=f7fUvVTpkrEP~;<>eBIQ! z9kYb^vl61DAn!|Lb&^7h!$t8`DOmI2V^<+dIJJbLe3G2|)>oNQu+pK;0H0Na4o?{q zPfUoK#T65Gg_-;R5=+jt9rUy@;&~s_SuJS!OY;aX>i?D)VYr~Zqy()Rm<8=Vo=gjU z`M`xl0>IfAfU~9#2JrTq^t5@xu&1W)L2;RJhb3m3tIch1 z#=GBRZL?j!-e_(wEqUIo^=K;yx4k8#!09iF8O5tH zqf}&y`#<9RLE{&lTwFcV`TXkkB`m-7U1K2<$}&^N{2LOsl{$R ziLcdaV%COlPIFzk4%KAx?s*IEo`QufoMMk%Ks&Y-)o*NDfyA|zg*8Irg=-8$m|KZ3 zYB?#|#2?1V^NlHyXOA85h93?vhwm7g@=gi(C_I%LtaM@)pfUOsQwEJRvk_F*%M^|# z6T@-XLD5Q-N}*ytm=vTv9&vYK(Ys#@VDt|X#BxahN1J;`0Z>m%G*WX1G{fopUd7oK zeGf%9qq>H@6`y$QD+&egtcvid`7m7&Vnj}9nF$U$Uk$xN;c>m!j`fX3l{eFW2BHq$ zOex^U@9ac-%bj{{r}`W|s6J0$VekA&ni8sZF#X)=?*&l|X4Rje)*D>t>&HZ#Pg+~6;|N%+RkT|c;c=LbfF}nl*sj2bVSOM=~u~!*K^1? zLI2HAg7zv?ak5JN>FoCo(w{yq&SFG#+7^lez}WP?Yp96=-9}rntXe@$aRk4LFBip^ zkUN>I_Ize3;7gYX*c3*yH9c7~fC%tIg$j5#Fmia(@_T9_wO_z$2_^Le9RB;DSL|fO zNdE(pOVp(5-*JlnNur10TjGWPRgzW>&RgQ_*fg{-+QrACAiuaBM;^c9$nV`7C*N+7 zKS0QX4NUiue&hnwp$8~pssb}*Gn0ojZz*)FCvk+PQbOph4koscbG~*VYLJNK&dLfg zv1f5aiYXy-|A?zij~oLR-eqSsTNOlVZrpa2Zxu(L12&OWL-VFjp=<8y<KKU4@kbWK+=y~ z{Tk^rwWK%0D1 z_%+Cy$f(_4?)`PY#rv3Sja^;4VWo?aqQ;s^HOMYqpUKZIZD;eD>{>2du1wEUui`HR zg^`06PnXsMPoDG5&@-+Yuz@t9qL(3G31ujmEW!?*(Fvz7=7hX2Yn3+Ag-QWZ4^f9e zClbZeIG{Q_g(90sR}%RxqI!afeDgl=0c_u^DE-)OHjbJBXxmOEz+`F_#GrEE2ptX@ zctULMop!(}K7!mBRn^Y`v>ZlD3Q;JlH=sg9fb5-X&9*A~MGkWxz=X}U&QawXM@37} zof0ECOeRz%;#T~X3Si;PAKw%8C1iPxEeTm7RmiAeYHE$c`@Os55^Le@x(o3AKTik; zm$Kn#^SDA}cRXqVKWR@c9E>zSAQyuYkovcl73_=pUzn2nXdr3rCxQX8`^bFt!m6TE7ng?+p{~{$I@8tb7 zADJ?u+&rw~}ITRfbvbbiukH2+fNe+K~ijv%ONOHw-Zz!1J zOn+qG82+g0oz!~zIk(h#H*H(5e(3x9ZCn=X+%s^8n?PP2iO$oeO`eQYwnt5@l>2q? z^Key9lijK6wWKR&tyzO_fh-c`Y=D=B-4W!jSZzYZp}^KMQf;6(LDQm$tlYv=XQt_p zMQx}QqlHq}&|0=kayXT#t<}0p(K2oxWb24bAoy0`1wv&8BnPx=@Rq=pj0vHlUBW@a zHKWRCfdRdo)FtP=*=ZGwWuDETL>Mbr%`Sxpldp%mq?0zDV|KV4gByLFig7`sU5C=R zK!SKQFP}^OMSgM7jH2CmxB4Hst=E|eQ>S!HKnl{Ar>93qXyn` zV@2JeaBv~0)t-k?u{L%n$aIN9^+}8v9i|3PHy{?At#FaRVW}GQgnBK00Ug|j(=U=G zwaN;z5oy_HGWU85qJoVm5(rN&7Yp>7`Y*&%wNM-(^cpSovPF`EPp4ZY_005LY;Ca+ zM2l9fBa)-StDKNu6tZSn8na@k^IR?cf-Wqd7d6iF-g(q#Y{tLm50^vSW5AW;Jqayr z;7Vf|S1fdv!xm@||A=p@P)!jyD?_TN?74yqH6vgq!$uV#k+55qMa!4DXu0U|=@++9 zTcT0K2GXiQK+5Lwr)bmKg7;Lt0Ri$JQZh35qZoF!YC#d^ZtfWNpOt32acnw+m&Nw# zi&RBDynJj8_3MiwP59UqoI9goyRvqS=IJGo)`qGj6oX*|J>~)1j@ard$o6Cnb z28Rm#^yJY@1~v%o2hFgE@=@htfD?~Sm`e^e{Yb{c|In&_wNz>MzVFwO2F&UK8x{Hp zbLL&YuO-lqZ`GR&opfMYuwO0{0y%7x%axM+0v@3*>V7EXs?AEnjLH=+sjcXPNG-a+ zUxT}lgXd{_Ob0Z9%NRZm0&@u7(2HI$5o~a@*lMgF1uz_$F@Frdz&|*8S;NF(I-9ta zAZLI(x>_#Lui|QW?dWv=SeqK(>s=s4g}4W=1=nDmdT0Z^0o@QUHQVdh>nVC2&8Q2i z`rm}hz3p!yGr59@byAYn+fY%EPq~iDQuG-*%uuKEUaM3jeP|iI;&uY+SA+zG^SNVB zpE;KIj-6T$ihs zmuDZ&zjvlMlP}B^XY<}{aqfzEyUE8Yxi(`j9+9jFYIwxhb02(Jj%Y-Rs<_jJ3e*yf z7o3W$9Gx`2oV=#x$PJ0FWbh4%e?#KmkoY$wK5bXn489B{zP1S2pMD9d{-9t<6UP@F ztlI5sz=G6u@j&6Nk{xsj_qFcm=%ZRQ)lYOk_)4zQ8Tl5xEeDT>il=3W{rF)BT z<5<8EKs(W6m+_?0^yaR3&qE>aBG;O=C%BrRuyT0g)+64rGg^O!9-(xxV>{6y*ek06 zT{;J#bJ15#U7mULf!TT?z5k)d`i}0n2NilvYP$1(mlSYbj!T>ZA4)Z(gF}>}2|Lp} zi~k)5tPJ#A{V~ZmIQ0b+6lM$ld*54n$X}g%&-+&99_N8@49Jb&Q-2KhTelX_<3}HO zuug9?Qp9|%$Grz< z_}47AsteuP{)jnfwwlbDCbRX|*nsJUx!=U?L|=`@v}~QL>k)4v5fOB^Atsiw(yb~r=&2ORZ6w=`iQcf_Pw(op1 zx~+Z_8lgXR&0ybaG@@IjfaGe%RQ3lly_JyZbuH5qJHsu!Gl(8~w}z|q*;G~S2W>ICepr%KMJdOLo7iZ{!}v`5Eqg>xMl2oU zY4oGgF@!I%`h4%Bi`JvX`{31wM!=Xd3FhJXE3)yQyUZ3X{`0^CdE^exJv2lAJRp8O z_?|iX^`5zh=%4q#hyQtaW(L-;vc646FpSX}y*!xr(C|HZ@4b2NVQDDS6y*3Db}wj9 z^4`&Q;izqD&*a~i2Mn`C8ig!y+r}0i3TYhMmbNvup?s~Z?&s*bLK){vCr3Q%AV;jP zxS@T6V51UDvpMUjKLf5CO;4?^plNlL+GC|l-SnsMbC7OYT}OjHI_WbXd04NOpOF0| zL8<;g^7{!S&tD=|f#SQ7RfAIPTk<51iXr19PMLKONYP6v&oBH<+jFT%L9(qSiKrC3YzvYXjr=4nKO_df#^Za#H5s4A)9 z22gmS@HySwBNGERP!F4Yg*d5%rUxIHxQdkG)Yx6m@c@(tc;84wunx>e`~VxtuWM!y z*9Kcz9k*CSRiO#7XK}9l-9%4=B`@wreCDIenwJz99FKH1l^L|iky3|)$+c%o5ea|A zN%Lo^k>*YpQ<~8t_xmA$Z)(6dO;;>w_U~7sV%Vk9kS}!SW#2iqGj6LNyP1_3HR)=v zc17_y+B~xR^=HtFCh27$J*x+aXmlkz;<1th{fg8GdSVoW(QaW&JWPg$#4 z>6GwW#4<+j&p@;wm{&+_iuL6J{L3>Z#J{uN=5oc9#u&A^#*%sWeIRvmYqH^WmAxT?MWBMCLx=zv5r)%hwz{}kPDX2!h8jaByM~=v-nc+7!e&7Pd}l6J*6Ym* zx>uOvQ=Op3`!tsx;m~udrYbgd{e>T*SgNc%#TpC-uw%{$8d7QK9l)hzIj&2{*=mr_XH!uqEFe?ra1xvz+>Lr;p|HE}T@V`<8j?}{`PA6xaLgvRuu8me zfjb)0SV%$tQ!C?U`a8Zh{vtJ`T=sfWM;dMmS_+&ooy=#^e!fI3#5sirdl{$T){`ur zSobQZ%M8kVuL9J-0T(IUbqNZ#*3p$3uISe1t)^u4o&)Tjn8H7U3`lZ$;F^VJe%5HI zck?HfZCHY*;}nk6DW_or6~S@;LD=cy+O?7}MDnU6g|;`mjLAq+ZG`g@76rR4ISQ8( zF1({<%0rM<5s|G*MZ{!6Dj6l2{`s&-wBQRi5h%Bct66X>oqby$g(#y?S*=#|&Qd*e zIwWf$w3Dw9FX5*4HtbV`*r^NPaQEgq-Sp1pmdYySqV~EpFH9U+8UuM6@ZgyIRuKeA zMpvI~6B?bZJE|bUmJxr<(F4+SMd5alr5j7DVw=j=mTF~b>f0s(E`qpFj4D9_q36Jf6T19#yN<<-KluHF%#+9oL* z*+}{((uRjVX4lm3~pj+K|bkpFyPS8)@_YBFooO=r)6Dyd`mtGuA8Da z6~?w!xr*Bl%grWR`oL0*0ryIygPc{0_J_xKp%zM7viyP#qT_SNjAOg)PZ;~4_JYz@=hEY2-X32eP-h&|OJK^nce7|Wq#D;Bk)U5-S#QMY#s z=Fz`#L1tK6d$$k?B{;55YDb5-g7 z_)F{gd`k3#sS);wTUj&XMg3a1fFG5}=hD#yEhy9KaT2VM1*MeTQB+e4`fL1`Fg z$g2FM)R<>9wdx4f0c@-%wQmdkek}&0d5fjKtCbRa7GE=;PYrFhCtr74L1PnrX~uU< zhVh9gPe|GFe^8jBrC*Vkfy|ENvHDY)R<1LRe7vPKM53d7^7Yvh@-_pzr`Mq-ISbt{0)auEm64C;Kl3?#Kxl|!+%M5PD1AX_CfV5nQ3pAe zxk6{!Iz!EvHr7J2pJKi;2NRZ_kr9Vu9joP&cpAt5t%=@-YuBIsOO2=kg|%y!n@%-Q zN14NU%j(^b(R>!&1#d$^xeW^1pJu#v)Ij?c^tm_OdKMGoa#YJzHi{eX#k1lP54e3$ z6V@{SgJ4KZ96Y zIS)T)CiVgT#NytqOQ5Q$&q5K~9AqujYmKv-53y5*9bAm89*IZ{OpYneYi>w3b@KSW zOwN)QksCTr`%4Icq3Ah5(KGaqxKdYraWWbPue`xj8^>*&#=G#zks#-l;V zhZW;~4p^BcCz!8mA_@^CY9afLEh>5guH?IR^4=@L0jjiRU%v2iD2OOH^#OD;rlc>|YP_AyKiE@6bvgotGciWx#c1U0;({LcEahp=0xW!9p4 z3gqCWSfgvr+BLDj7(;p6HllFWQ5&&St`|WRc8WN!9^=mEQ2 zuJFn<(&70_e5QKHy@Gsnnf?q|yR9X(ZD;z>UJBzXX%_9H6 z{BS4y1T~tMw`!Go?y^c!pG5#Yn$n9+@#3NUL*m(byHd(sYIg&x1*feadj-++@*@#I ze~gB(!;H{Cq|bA1dL{!>U!XN_xLMLW9E3x#c6kWl=8o4p#a0E$73vQpYfNW4 zpn>2l>SH1@bdoec?uAOgi1UvELH;{FFEPGudPx0b@rFpdA=2u8k|kx`jT<6u*nV(B zq~)+JzClEqRu$Q}TpXf#n;36rcGOpDHB+5u#TCtg*-@_^eUr*1yde8Y8Uf>GN0C(^ z)LsRTWg+|uB?g61-DHB4%+lm@Z~Mwzjv*G8Wdu;%U=CzaMH`A^mcp*jb=&le3QSX= zwd~NuTC3=7x_FoGT(p$LEJy64KDVl`-QjA9&?HH|UQ8%8WTJ@FEcA&)Tof>~0-+Hw z&)XB788d07~Z8d5LqxAZ85X!ik`guRCm7@e=();;obqyUZ5MOT`--a`k+}g8W zgD|yY2gGtO~ly#gGLZa8r1 z9o=rHz*WrgR~`?|G`t=EYp`{4542_D$iCh;eNYQe!vfY`xF)Wl6Bpm{oZ|93N*$(+ zNV!F@?clbEG6ky9^$Y$gqLC=wisGCQLMApban-M63-p0(Cb7*eW4F{hl)B_jqZ_Fw z;PBr`EM`kC#_kl8C%&;~x!A$OQZ$6|nw^5U4`^}uo%de(Z?YzTfsMWi30lmc> zgkYJFYhBHQx!ha%t5-=?!JI(=dJ z+NX&*%}S>;jX1B)2Z41e((UM{^`bSzFw$-D9Y`{L8se#qGBzZ=r1DI%t<&tln{1zt zkVhCbr+Tgh6|&CvP^J^|j;@=_^7$aV!`IK?WbissGEW;IUBn~8+Gzy3pBR=x|Sxk3#lRTN5z~2RNi}3HSA)4m`^5psrF0v6@tEUGNu6W)mEG`8iYpt-}p{ z-W>Iu6<3cU==Dmj-08H!M^2s$8pVyu)e2HRD}E6rS5MNflk($bLx*K~Zz(>W_Hxgi zKRXRuqtaNdwBQTp>w^4B9WWr>1vW{Y2CuwAK6KZWImkYxSA=161+|6AFHWe6Yv&Oa zj5mnhZ6Ssh{UC&wo8tB#9khzgMVEM>ZqTyS#r`8rB})IiRw;Ecl_n0CQi$;$5xuBY zQCX|3Qh5qo7#(mdwW)NWH>Ic2;t@+Grq?IL#F>Pch~ImB;IWmqchX#QtbfN>-E)af z`tOO+e##2QbMF#DUY*$!x-`Wg=6!rjFF4Dk^D z%*YFAnD{=VJ*O7h)l+x8MZ}Jt;_fLQl@q5*#fs${ud7p7^fTMuOQ@FMzd-0 zIWSKurp4*QH>*PlGtI`dh|aMGMC2GkF>kJPe9~2+q9VkoV69&c4nlFVvMOlgU*cGpT*MSe+$n4cZp?&`&{;;WDO!^tqWnOvZn-EZJE{+P5M}~sp=mm zy6&%OVg*>n zmf({Y0=Ddxi3y_m9ASmLRxC*>!_*E`xxk>$0>r_cUW3K)I77`2M-yF+wC>s;(NtC-@w&TJFEJvK88ZIdfynUHXANm4Hop3< zjlBBAhDkmc*QtY7;5CU*xBkZ#)7w33>`OJgCNr<3$I@C!ANdhI$A;k6O8Vj@>|5;1 zfg{PA#Ba)iZ0ANiD8dW3{0C$SX@TM|oCH{v69BxqAnQF749 zgYhEjhc!_AleSaZfV>=+%bzwZ33#fwBavG)f^`wF86S?XE^-Dr0|If_k7ISc_`ms$ zLIZ29y>Vc|IzOL~E|#BSgoE0zOng01Hi@E`zoMB|4b6hd<=0mYv=7B$zdG7fUOzU| zD?54}2X)NaaOwO}Lw6%A;?lX>S31uZVye+=BAmZOqw5)D*?cW6+5GYUtl8X<+sNj{ zck;ETkaxicyIiV-WHBxp;qw71sodd87gT(Nr6KH$Hf#(JhX(CtYwP%N1bg5lq9#ue zP*#QHDtz}WKk*UFmYzFV^ByjGgG(Jpy%oq|GH}*9aMg$17R~|cbU_EpW*Vs=N0J47 zN~r#i!vlEak&Z~kBP*{6u-T8F#}OI4^$3Z|p2xwkI3_;7JeN(j24EPqL?tA#{amP& zP!yd^DZZtV8q2aC=4d%D$~ullB}>ST_Z^nhHzq_`Tyr}dBfnDG6HhfSMTr;a{@WAX z50sS3diun>_9}B#fyct~p@H+@SfR^4F(+9_e?G?SA!ZibWH2;+Uo3`=mu{ zm*9@RmK4gEFyQniDh+%s*fgrFJjztTHi=N2Mk!AxtL*?{5Ve6R531J0(KFRLrlVIJ zY-2w_nTQ&0W4E3fx%i&l!C@vl+HDUY>%U3xh@~{f!gq#;^n5UkiGeqxX;A3r8wHmj*Wo@$1Xm0L29I` zj?a=juYYTHVBvpP0@vaE{?mrv6%h=-+tq~RZZHyp16=e|A1NP zmdb0_?;9UbJ#oy#r@$PkoW^|StX6N$BpM?rx*38-Ybx}&hNF6s7<8)oH^$;qinRM| z%`s-2aeipmj59lkaS%)974V;Q*vFh4{8a&0PaZSTa#~FE>fd!ts9~bTcfis<53eRd z+vwsX87xRQA*^;N>xa*Pwh$jJ5`!99Wd5=(9IC)Fzmdr{mO7O>BG6h{xH)m1_NM%B zkv=R}P!|*mARQJ8bDaxBiu`h@aWHWGOl-UsK7)q1R>5`Pr9_ZhmXDfD2&a8Wg$$*a z;%6YRt9;ZyM06oy;aVKTjVLd+c_e|@QWNHA7%({yPHp7B! zCKjZn86J?7Rx4Wnjx)w)qKDx-cMCj$YkeiHGeau?szFkds< zersybWelI0F|aXe?xSjS3LmYN8iiv>yv6sW2IyY0Pb?6{Zx?V5wR!*ohxD3K9AQ0) z^Ta1pL*}4bz7|5}nF}dHalx>Q+X`I$I}YqGqy{W*u<8{p7U1spgOM0g#_Kmz1O9MK z{8>;ZUbw&=$VgV4+5st4t={z`SHH=*Y$fIrzvFZH=c$q7?S19F88RBsXfQfc1NIEX zt^?b|(LQu*+#=h(44xf(!?MV+i;rDo8kL*gHDs&QG>V3`bU-oOq_MU!qYcM3)9P+C zQ=AH17dc=;JEh-?+b>_@P!$mZv$b8|&AHWkX8w8(sR_(m?FlepDrUx;Ue)@6A*|65 zqfbA*Mn)=WQu3I?Xo<+HSm$GpnAZ1_QWy1Rof)LHAwFop`lRde%$z$6ZyVkLqn=pjzP%c&9Y{S$MahAf)A zFUKSe7n~MvnDreOoMA$6zC1UQH^_)i;~_H7UcN*El?5luQ+o5#w)}1E^DoKRdQw=F zcG421M`!dhG^Dmf=|6m_f;)W4B$at`swg%pUE}zeEnQgGn&$#Ptl-oba*m8-ecLYu z4Ixyh)*!=m>BLT^zsyV6vbEn4wSIQwQJ9CvyE7%vNx-F(U` zk&0@3^o<9pCA{8<&|z(UeM0OQ$zS~i%V=&HOql(?E%DiKaXHER7q{r#6)w61dY)OyiQ#cVNWk>}j3g_JB z_Wb2pq~g;5&E=U+_^WkA9Yh(4Jbzhc%_B9MDLaclh_324eN+p4zFoA_)WYEQ0eC^>09h6*MmT4*qqQir^AJYMPEo{|T$^0X*L-5cRNP%sRv z$TI9pBURpjkm$$YIGy=TZE3K64WeSMk4X z+&8zc`UI_jsp|Y#3xUq$KYGeB_>j4n)8y|r7jtGV=7!D1?s|*Q#j|P6#mCDIa#(Y5 zKiSG*leZ<>xoGd;O~uPD%s*Q|{5Aj5vj6RU+gqY!KqJiQ<;0v?`tCj+Q|Wy2SDZ4X z#Mi_1*H3)ACV?87MdN;nn(jYRPaHrF7XsEWhU?>u_d-$#fWSjpFNa3chW+?wf(29W zSCfPV!**KC_?b`XnJ^@>m~n}ii#U+=4RXE6`aH5&K5D6w?>ftTX$l-sQdhH7LP?)? z=J;{^s&tS)jVg4FP*kabwLqJUgFE) zjQ!c)*5p*fGGiaHE-JNXVYHDrY~5oX68d!dl< zf2mj-yZLzt3eP10L)U0qX<%NCD{qUuI9$U`SA}97SS_aW z_HZ)ejTUMfX%r2*JlRLm+E!S>pba1QzhqAPTZrRRfZ=i%aperf2m?+6B(bt#s9%&N z=S038mhv#TX;T~w41@y&0xk$8w*W~Bwfg1h1#k>;@*(!lEK=q`ng-ITWv{Arn9ryU zSRQi3;t9!zA)E`iLA!v{5HT&{nmNcIzCS%4nz)IxZR*m82ZJOOB@D!37lmgVWzIt{ z$0scCjbzLT%k(3>Ea;5o)WjTPv3LrdVP*VzVlBD^lB3NN6LPMF$HZNYjfE93wUELJ zvT#YIyGigllg4Vc5K5=NNiaH_ifFM=N~yz1a5^@kptGPJ`4bmhqdDyIZywiS_nT%< zRx*P3?3AYUy<<;?&izrc5M4gEtd#Q24ETVk>E&TeW*9ggW37AA=oF7QX4<6C2zi=Lh+{@gLbaGlx`7e=R!dzUtFJh(7_RXCJ5 zKW$~ryS};YP1|?V$^*jdSB*>QD!Y^9Q6QHst5KoB>+n}>+nH(CYv!Ap`o5@AQUy@&5TkOy;|L1S2<=#lD*J9%IZ2$i{_Z%m!2 z4~=*AM`)Q$ReyM_)k5C#KRw~VG0s+t%vSP*!o^nNKIC!v#vEq6b^~YvF=PBhT8#0g ziw>e$Zp@#TR@4?uQ)Ew(a~*}zDa;Hveby?-$2Z;|s$qgS6)G#wzE0d)@(Qf3PLqqJ z^NktGE)vW`mCJFST=e4HhpM)s3O$*B!cjyeTrwN@4;>$KkS{t`5WwK#z~D}01>taF zh;o3VW>rJRa+sb;10PG`rqdtz{a+^d-DQ9|+MFdCO-xz~p_Dot8zajYdvubC*6Spi zs2d0T!#B1&UKzNU7H^^AMoki*;%K1RLh>XesdZNxc}tt!8s2ZU;2omUL2x5c#fvtg zvru~od5hsHRU{IYH=Hof2OFJc;~1Pq>_uW_MzzLTv0EeOvK?gPip2^dcOv+Q?jmTD zhX)?q%M6sPVx$~X=3)g}Xst5*Myp{FXA|Y_O0m+!#R{v4ZVk@M+ka1%nyV*U-IbFU zTjZ7rPnJ9Nn)*>#35)c5+PG@pKPgcx{++xpO#K%|Sd5M~3vp`exq0ARHr6t{XCR-L9-<))T`-u+^>@1XO?%32wclDp#-C$^J-5Y; zR+sp6v^gOCS60{hk)zlCA^=tHFQ_W_>8%l6{QiiMENZUU!gf^$R1MULR)3(WqBv2Ak{<4GYaRtJrfvJ6j9n>Mm3o1?8(Qn381$?6`$SH)#8{LL-1rw0$+) z@zo`6rf>4-TE#eZ-`b6?`nJ5>EbBzy@J{Q#voO8yzrw`J0KL!YJKoE;iVNP$x2V#y zV1F3ux$^DmL3Gzk@RH+@6|!-`M_JCccLsO3s3VO&Z}qNIoh3$`s-DC@gEV&{{cV74 z%1msYKe2SLZ$kuIh^xIlDv0FKAJ`qJ&ac=JI2xWH`z672F?>EO#^~6!^xS zE^u1+&O441aN{O_wNjTD4}UR#_>ww&*V^<=CoqGV(oNrB2TptqfO*53eE~x!ovoAjNWsls1%U z1eeS9bP`@^EO$`7cR7T<(edJ+$y<7m{UXM%HT}}ESRKn^_T%lvLDI4e+YsNVHFH51 zV$5Xj#Y*yp9!Eo{l{Sy_y5s=4=}cyEnc>WP_yeHkQ9xZjz#n*kjHx6dIRRdgy3fnM zJAQjT=;RR(Qfk(hsUBu9KQoPErb?relVcRavfm1y4mF;}MGblM5+Ot-ll|4zF3M_d39C%??wRf%@vc8o zoLk#IS`^_eIoUl#a66y@h*p9^)6>(Ldnp<@gkCJFB{Sh;OafO0?w42%LY8%M$}i%w z`;y--U8E1&8UDTvY~#-{OX7`<+q9avWqwz}SF#aiOT^?TGH_FXJGdnp01ZuOixdvZ z%C1Hsu!FXO=n`%s)hNVPLu#7Ko$tmHR-ABS2~#k_jV0_@ac|Tmtn~u6_DVx^)9Q2B zt*g(|e^#)GS6A^Ly1L{7{)HOhYgJ3BBJseT>N?u__Wix|@pJU?^fO0)K)Xm@mi;8v zD26T;gFiJ(=EXTPENHZ#qE&#jA$F_u)(I(4r(H0R}? zLpESn-tgG_l~Jbs*Bm}ePkkK*1Alx-&xg3v)@6!2+8mH{wBXf`9KFUP&-M%2D=SU? z)O;VQC+O*;3HajNo8()0U(ehxYQ9ylvjKkZD3fi#7^%4f+u^FZ1)y3Gxcz(=s`~fs z{c=;)rH$7|GnYCdCECO2AX}sa{p>maDjCbZu$)oYvK~kczePG=t5iG#h3XtVkE!Wc z__c1m5zb3}eJ$`i-FC3tK*JDbz*=P;adn$G`9LQ%sO8-ivVp?f>_hpvJl=*lGnvoL z&gAE2pzp#vjEm|iTyLH%f0*)Y4P1^{1K#|+H*+Pk5;V|uSi{*X83YQ>GYrMHe=Vqy zbwbf_6>LHjS^qIeT zxpIZu=QIYfhTJO%QEQcbbUssjsOLk0&KO_8-dOHbIyF5ZG1>zldkAX01wH0!n9G6A z0C?s~ksfaH4aT{n=fuRpTwQNeuqgyb?N(c=RU2nah-}`{Z)?Fij!tuzk3OfS?r7dS zdYb3vC|1*vF7V(4Ydv zdJ|ff)przgfGz25Zo1fRb^*mKnHsb?(S}wAMs>es0xE^!Bs<5R#m2bU`t%60EF8M= z8@NmXb9JfRMSo~bCMS1%&98OA(BP8_>QYb#A+9CBK?C6e6v6*F!`Tnz9|pmJNKpcm zBhzHQ7b}WC#gOwfpiLKLklYYdHw4uUL3Kk=#q{wvl%UdjADdmDfZmz3b!=^Oh}rdH zT5VNkPF!&vXm*|cW39gaW7$s@7hnbs`4P8VTHthH>WK z4?Wk1JnQnLM5Mg{^N_Gvr;CBOTiqQXg)S!JT-UpmkWvXKFhv-V z23$*(qhlJx%9JbcVv^$31VC*s(Iu!v!B!O8g8o}g-fWv0q>+f)r@DFF-vU(#F!*rk zARiks{tE>IEm8RZ5UP3$^0;6tgECY?S;XkHtIJv2jYL!%1*@*AQ8#_wZ;~O9vFg-8L7F98`h1#%*KyFaM4d!Lq)o zu%<|+b3GpEWfZ57CW+j#rLgT#RercRxneft10D2R0fXgA8})Rw09ipDRjTLjd)WtO zWNrTkX0oyyvo$<1AzT&{OxzU+|JK`;54it!;Ar!{T~o(`e}Yb2keg(QTdWySp3T@Z z&6})W-j|?3Ts^ckO$t8w5S;N}ZJ19HM?@7R3OPV+z$rTQ%d$yD{5G-5*f#?i7VM3RRBi`pfP&1C#iu=cv)-cW_hm|tmx znG3!L*d|GD<2Zy1KBkGa3oiJ`w`#h5r|c&QN}61-_!8{+GqTisf9l{bN-JR`vvip(5&uWLUzTp)nOBIKx!8JZwB`y=H3B58#FO{GSJ@~G5`Uq7A zq=UD=qxBfJ0W~qreS)N)|KlaxX5NnkK=I@S!>fz+`+};wuX+z}dhi^#+xP zg3)XuoZ&=*<;=*1xa9~X3)^1q@P>9#T= zJ1=6f>SKX!p)8?9_- z(||l@Dl-=iti(%7SvG0GqogdKdVvZ_$eInJN-mX%K*lwdmf%{Meq#$U&0M7c*J~i1 zlB1hz>&VEgbjo#Qg{RPyp8e{5Z(;B7z4#|db(;amCrqi(pSbCeItump6LSo@_*Rrz zJyokW5!4bO7-$R8r%e=lsgv)qF45vFQAP{OBheiiBr0Yge*=oJEd zyzC-wn4~+xOF_-cu8U$)({$^kI4Y$>P!&%k<9;$$rH(Au`&2bSB>XxVA3xBJEmBB#$S7MexROKO(Dl8F-RSLM~2t2!@3<;d1K2_BL z#GDAq0ZXzwR9Hm`Sp-qjCwWRxr@0YMRGiFFU=eP73Ari{<_U4pR0M!^d-PxLVr0Z7 zi?}LytJ=*?R6`*x^(1nDQILduZ?dKdhPN|n7=>;fUKUQ?KN6iyg*hx2PD&lZ3p7is zU{CpGRgS@x4&!KJF{%ZY{nXXBIXvF8sMa+8(rc0>@}M8Rqv#6kY5)Ifgcp!IDbF}R%zEk|WNMXcDl`7M?Zu?CDicJ*)O zb{~r#MKk#;POx@r1iOnDEZx0pNea&byibF{D7-l?Ar&sk$8n^-AvL6KwI#BJOZ>SD zg;8hu-9yB@HR#Z$E;tM~m^CA7j+zH?^8G++iSgG^WuA3;vfcqDospRw68WmN6M z0%(FxT(FNt@GqoB@Sbv9&TZptD%zGsyp%e05pq-}c$7nZcd7en;TBEQjRSsT(e*2g zV<|*rtBArl=CkPicHJ`6X$}g~AS;kTMrE{MkQJomQc#vlfeSDRgUnROI%W6O-@5BH zgN$|8V_zCuYZUwP*oEL)C9Lq32NXW)@?23!soKRyZm*}X`u026R(u=&8O2uov{6X! zKix4y(g+E@NR@1^Wn!0Fay`-P)^a851u!6S z-jbck;(8`8I}01^T(?cnxJ)j(6ZcZ$l874aA!BO`4>?pU5z%b#6#5BdQ z7GXhcpjKl5iw^Md@xV$2<&7$ViDZ25E>X3lz^=Q%y19=gZHQ^(Xmdc)*@8kp8oO8J z6@poHNAJ8@L%80$Ml+kh&6sq>()_n(S2m;GpTK^&8TBs=*DA8G8FfH%mIbYTDMh3P#Kzg_Sf$bQg*Y}S0skX2HH?15dZQv*!2 ztIu6{k2Y^dqL?YDnJ)g=)xW0eC<^1(r$*i>^1}5IU=|q%HglmfD!J}*3rRz`J`EgW znk~HF&Tp+@+IhszAFne0xeV;{^-AM9R{%J^jiTd_B=fUhv$)*g;xIv>#>Y+J0;2TkscCedl=-^-HCR9=#c?G5ySEj%k;+VHQhO zsUt%ZVXyF~CYb_$WxOD*MBFw-%7D(}3^w80$DLOkCqnM*yjp?0vkLd@GPD#5qmaL# ztnd`l{ZsqJu-4jCPp~9Dm5>Y8t&ex~!l~H8^(s!(Pbc~vzFmIyc}+b96mOT0PCy$A z=ZU%*OsE6AvX^HRb@cBzNj{qzN%rwnTgb@=BY~-J@|0jEH{#z*4NO*4E}q0o#(ZC^4HW#>TQf;hn6eie#cJCQ_*>+J}?odQa+%B{GU7b>tFO zw7!-=pQGxW*O*QjY2$repLHsjFLa8&gUo%T+AAe$m&*BS+?SFxi0TD${(!_v; zQyG{rs2va60|o~fP;2Zm!FWZnwgsrSf)rv{G zMf#r4+i-5$QHNcbEH=HYI=~_+<{rI;uK^kvPDvXnljZ$QUPG52)mEV8SQv8^i3u-4NDVzPtM#jB$kc!#2r`PWWb9 zdT;3~n39h*uklv+?nHmXrS`8ImQ%p6)ZQ=n#6nU%83=aulQ{ZSUfzcJnxUIZ4Z0p5 zoPu46C2X6AG=Cc~?1uyKh=y2`Po;*+K3;ehi1NW;@DAyPs)UjCB+e6;Qv>y&k_i^H z=GkC?V|=^lW@kK(gWgCD^tU?1mKiK^zIGwxFmep*Y3GJn#WzDTsu|+H;|P93Y6#xt z6u%aVeHyrsG~mZ(t>vCIJdPvvvDA>dyWe(UM#Ob+eYO6o zHdPu#!CJ*%Z!oCzr194Nt03hd4R>T4{MD|uBF*#|H8}d*(`#f~llE6C32j&SlSl)W zsjq6Aim9$IDb-i21*7Pkx!J;S2|z(;BDc1=LP{|}Q93DN!jwS*F-5CFmIkr3Q%ey& zF$Q+ldLHXDteVd!1dru3h)MjH+!7y;>?*lw@i%6NuG@mt+@5x2llT(PLLBd#1QT<}}v z(7Jk1$*p69T3eUlS;BbK0j8=8Ls&<-zV0%#d_R9;am8pymg*4L! z-|&L9wF(8Sqi6;SY0x|hp2TG~_=T^eT0(DxCd^rqDA+~u?ueJB;P7jKyzCVIir*O* zOP-p)LU`&G7k#-;=v4K>*uvy(@dY645^E9MM?Ok3+u>>%-dhSg0fQl0AFR|sMuA$> zBevF4Qe7p0aa+jPaKGIJcWg&qY`}GmL)N6-^zwDKtJ?MJjbLLtxs{63R?`MwZ^K3s zh7&6MuWF;~HQnH)8TRsZ+(^ERBjEl>Fo4_Af9`^FA~i(WgQLKpFSx!z`7{i8$pRx% zUN0%uoSM9s^;g+Pwu_J6kx}VZv_JNUhdU9?gN;en*Q28lM7Jqv;C`tgt$tW6{=YT$ z;-7n;1FG2KkNzaxm2WLTcBELplghSduiefDHs|L!EV(grLo>r@O zr^spp=aX_qawEA-C$j zbvm*~RS|K}Eh1{5X}|sh>vd86Ihv%Y>d%ffS7dBnGIM39X(l1Mk8DfWe7M}1{tJ}~ z*UT3X_Cz*cAuI@c$Fe=u2hvE4f1ocd?=?k@r8bHwI(@ob38;i15>#OIP;B=QbM(Zp zFWe^SI0H>pe;xE67t?f=lBy_>YOKIH1uFHu3m*zDR88K6x@0X%TC<^&BmUH z4igM3rb4sh6O;C1xa;=?cYR@O?)vIDo=Np$@#~<%zZkS}Q>{pE=G<;v^B1hZ%7$+S z!W5P@W;ZC5Du^nf5F8www42RNK_)q8aLEw@=0O)MSiu9D%WwuxA^6hKL+d!sBpFFE zQ4dm@?#5c@g-`j@FsViO%L;C3AbF5c$_vMW+0w$%M;hkzarPe*3{?!HUdCrVQou7@lc>7!6md2`eZitN?{DuI{sQP3b*(q)EXK*u5D1&6e__7 z`w@yv2!&xiphDSP;{^LOHf6;0Bi(|ISV+mhh4mAz|{u2_11AaBa|$4`7x^EgrP4K0f-d(vcM%D@lnwe zRb_ktLA6*`LtHD0Yr52B3#jegERnheC5VyL*uhyE-MkG+MG@!{r(aMG*^np?)dD=)dIXyW+t@WNpZ(x>eOe)=tjg7TK@by=dbD4!wT0KsJ)!9^--a;y+4tI}}QM3+Piq@t9k+qe7 zkYMWJiq^W3g^Jr*(Hd#Bo6)1u+Pm=Am_>WEdm4Xf>CTA#cxs3p;*b}XPkM;RdguVr z?3?|zTZ@XcqsS1#pSt;`o2Ea!v%Qlm?~6`6_7#PKcUJCc9^XIAR~|Oo*fq$^#9nbV z^a_Q?b%5Vk-)L-@V7~+~2JMB>e%jLf&QA3g0cAY-?I-a>Ym3+ zm1|z5G=DVs;3FO0HyZ7uk3V)2j}00_TEzkXjdTEC?E(BTdo~s`Yq@ZGerA3xw>fW^ygavkeQta07(Mq+ER}G}cad+$UgF8SZlS#GEp_~LwANGi+6@pV z*Y6J{*NCZ(0TJvtb_RR4jbapi=p1x{kfJ5LpwaBEmo2Y3R0-qSAa^@>3x14RtpGU; zO`xQmF7V(|V}xSVLtNmoMWGL*Bf*zT`I22!wrioJAudcuJvy%ahd4?RL#fq?$0mWq z$dzcaen6+ckV{>ewJsIH(Plx}*^>}Mb3v3!cMfcZOTG55Ianrs);EQ97F%FTqF9VK zqmQJcfo(LNEY)s5GEAzC$n78buMW^HZa=tH32LOxE8D|7&=6IR0vZs`qkpWFt-+fv zdG0@Y?=8J+ElxNu6Dm?|7KBhDN^W0gF|}>Z6o)s|05b`TX_4^b&p42jA84;&LoLzE=g%ohC33QjG&1(&6s&%}Wy z&jyOWuH*AHAE%|#urCE6q904}`Gj5Equ@OyYQ$pIQqo0+_Utw6-CW4P=ffSS*IW~U zfGs0oC1@9*;Zj*y-t!?1=nYEY;!u|Me0j$M@A&j40TyYg?0A9rppZ6Pl8utIpcHiG zRrplsG7JG6P$k6x<#wSytU|C*Z=C-Ou_9b3q?XWAI)s8yLN*OuB8Foo;2>_{ptrc5 zQZ~WVfetks>e2HMTFPo!a3YSD@W*Pci?9M8S3%$;2v@A4{%yBTRdi&1Tj8TIx)CM= z4p}1z;V?mgV2TgLU6*a-6=ejRtPh1<3?};~?-G1<-Flv&V>=Q2r&SiCwALYkW zgY{j5rLYD3u$C@t9OpoaeXK%aN1}y1~-Xkn8=XOrM^HkIwz!Y8E;C&B6ctl-eh>{2T?IQX2tYdX#XzGGh(5|eM^hFRAsyEZ*5;+UETK9mS>@x&*nFm z=cqL03@S=`uK%qq&&As1MZHMw$U4VRwJ_5jKvR@Wrqe^#(a32ruYN&0ZfKziqFIYg zmi@u8f*t3;3(Puaz)~3AP-9TNL?a=`FTs$5?LP7Z|LNgdl_%Ih8=K>yiexD(O5q&BRHAQdf+tNHxit4 zdVE8~=#^q(^>3e1ZfTg*~ISoT}$*yvy8b0%;ye$1&KTriw<48?QvgS$ZBi|PP>_I`e%-{;UwFerZxjYIs| z`?=xxF9%fgq%-sW)3liP^shP?!D8OK#HoSYs$>p9NTzU85UZcih$dA7ic#VLZ8f7F)!n0u|`{Gm;t__vv7WC+0WV=^789EhfB3ul9 z4La_nuBgCHg#y%ZP)%pcbk2zb>I(Y8p|dEg?_w1`G7|@P=d{N~dfUZv7KkkCqD4>(ZT z@Z0S45P2(n#w1}smARW zJ3-K*nidGpqI(IIF4y})Pg(W7#EDbYhw#rR=M`UL$TG*DSo`;HQ=HHW%4)14^grNC zwDq=+Yxt1fCLw}=L;e~iyqdu9?71A5_gY!IzPU~Jv}dyBH;e~VO}OT6>fsyY*BUNT=r&>W>{9QUiK;eOvp z9Bt?mjrhUf>BiMSyh0QO^zS&DUz-}52jo(;5YdlZpgKaC9J75Sx_5-7ZH+lna~C{@ zukLaR3Ps31Z{uYB`>B!DRfx&ZN!)2*cCUUXi`Kd%>E3fl_BLo#RN$1NW6RAAlBtm8 zl{JNQ-}f5r>L;n+sp==j!kqkCUuR^7fi3PfIFm6;efj!Tgw|8lWn{CJd2d<1irev) z*8O_j$Da-G`ZbgaMlGKvXUcS%WR!&h9z@KmJTaMRhoJi+MVrgua-*p1C43Cw zBJmrO{3nTtu_pPTLSId=c@Ur6FDCjJKDj$)auop5Ds~Xs0{c$aL{OZE>P2yYg+?wm z9V`mm^Q6!m{ew6e4!>Q9Nd?31$#7x}jW~N$tN6dpb(q3vv_tZ}6nbRZD znN${q#vrBs9HbmBfvhE9B#`+mx*d^RY{3sO&kX zHdXu;2m0$%19~qNd^5e{QhNIB0?c7RD)eKyFu9S}ti~xC{Tl~lM(+1B$E+D6Y3T~| zypV#)<1OL;!5~@>KIqMu6U^75YfU=l;U5B0gLaLP&7qI~jJ6vLp{%WCo29GYMlvnZ zO>G??n}jF{oY*C3i^0ayLY8M^9XkO^UD2Z&5-A&04vP{@d5;i|MM{(Hz#~jS*w6W^ zsH_!u=Ve=?kiR(&hQ+?gH(-}fCDw|i-5R}Yh@$rCL?6S2#3x?WlvDt)kT^lx>4=oF z@EYqg7t)T^tKd>9H}l;hj&xp<_X`8dSwRKI$nG9YlSS7tsWKK{zHR*n&a|*CF zb!{7}PFu5k@uiIq-BgakAp4xe&GH1@P`{ZTHC5a$Br@{pAc-K6eW+S{WJbarP(rd0 zY7bFiU`Q@>Xx{@vPCGvT1qD}wR;SBUm&FJuY$EJA5T<;5Ssh(v`}62h4_9DABd&W? z+aN?8M3IjYWNMXqMHFZVAJH1`UCzn2WJjL{V7D(O#F`}<-`0m&6c_h>uX&nxrkmd} zu37(6hP4%VEI5z2xzxh5Ukewa4sobPv&7-u{g!|=YX3H_akZja{YIiavVzDWNUC@5 zPK|2EM_;KHPUC##!g3&WEt6nweBoe0xtp(D(2h1IkEAvyx7yOp!YBUR1?i}CJ*{N* z?nR~rR~o%dg3Y;fq_>4pT0KsJ)tNLl6APzw`kMr!M|QunS$Is+%Oud>8rxj;E*8a~ zJ4P^?WR#=cu$@oNmL?JjyAdw<%}_CW}^Z%YaK;- z!cKSfDjhScDmijH#6gKr38l`W)yF&{Mr%|{8NE{cYMp?Fe00%eU*U9^`$xLOFWAv%&O8ZM5+Gwn;nz~V;g;?8%T*aIr!f3&BPBk){0 z6ty@rtscQce=D(>xZHw{HVY|CuZ2)5-HnZnwf~<@h1D#KQtEILoEB1Gd<&ygx|;-_ z<6|sBBQ?kQ$^qSQnffn`nc$MC(dP3_zBUFH8rya%evS=GB_PSmWSvBK^O>ITIC%zA z_pd@k!PZ07b`nZLB&oQ1>#hakq{@v7fi>Hse*>a@A}K_)A{*UfyxVJuplKP#9l|D(Luv}7r*v`cgYeWq8KPy4wXJld2Ao~%T{x!U zJ?6w0yc<2ngb!{829yYA5UQ%8ph=yyrcM*bq(X6mnuEyvRtE=@a8KEeZ5ZD|QHnjK z&!r;MX*M$=XoEOMs#Clvb|7$A@1lHZD0rj+DLsphXh;PUt|#gPDK+p=ff(f1A`yk^ zkEG$@ycUP0tB5nhu2gASc{fu7=E@RC1Dkh21*0ME=wU$8 z2R8Z#T6*!eG&Lm*D&@7?{p8%;PYPHx>fdHCIY#ESVruBT)giJhOq{P>NQ}yN=h6{} z7Kmx}I0;q@DM)7vrBu3`1fS{AcM84EqA@;lz-E-De*O_lT#{Om^p8Z}@f>+Dpd?_E zM7Aj+rrRB;-tdvo@eW5(j$~jf>?>cuVgz85VZDbc{8eRsS!uvl~`5pGDJP(SlZUiR%i@H6*kJ z1*I1|A`f2=xpgLMeXq4>9P9Fr1j@x3sH&Q*{)PdYi5Ri`N~aRDTb#Dv6A=i8Y1 z&LiF_`p3&I;OM)d^7dA)U)X-RfhVw>{0hfiYN|m0&x?37IBV(hII;<{Tl6{if}a^W zF3)znB|OT_A}k)~WcO1+`%Cm*yi@ME~c+SA&^|NtO%>GxpBeu74qkR_RW&pXGFkb{`JZ@g&_>kedZ zYv12XV$Q~ue-Fepy2X-;YAwW`#h1c6Q$zo7Oaxm{CteH&|F(qaRw(p5kE3xiH8gHd z?iCg$z3*Jej5@pTpL7$lAe+3eNiey^+6^tBVvk+@k0#eHr8c)oxmNun5)AgXs286B zZIDaDk{m*;e6xGab(JmEr>+SWRy>E!6hml`dA{zj2O+xJtChdqHQlSS(5f zpAPSjxS_Y^ylkl9~1+JyX0Le=O0@aEmqK?=&u z53RaB@||`8(7P<6WADGRL#K}!*h|K-Fp8_=wuQK>L-9w?b4{}9pAdUYRsR_OjN<6{ z_hz;K2mZu3Iu@_v3cKUS6CfNv{s@v;VC64*OF|~gq6*5;hjnjz1&ZFOFznLVA02g) zq*j#kvTKic=%r1?#{700VfW(1UkG+OU_>5GrO`zOsLHrs>+4AHve~s<2Q^FK2guD9 z;1Hn7_LPn9b?EZDSzOO6UeGVR%F2vQc0HgvI+ZwOH^UnXH z|3R+|2`j=k(=q2CRNgjQr@q*o42(8mJoo|KRRSxq204I9-bk(qQ`_a?@KBN^Y3tH>pU{Lpy zjzV5%0jv5Mr|$of8nAmTDKDh|K>grr5|aCM)eIF)zO+EZXX0O{HWT}W-fQuUdNP<{ zM&RvN(*<&FIte763^>)chJ&QSH148>e^C^ z)E&7=Fl*q=0FJ^Cwjm~y00R?#W=Vbne1Oc4`4JdkFibw|i^CqaPbR-%SOUNAIrrV| z)vfocx}O{e)PGc0y?3{>-gD1A_gnxpRNjMs{rDUHMlX?ea}~qiFfWdfv5%l$L5vdZ zxg+)fId^ajLfm6c%Vv@GJ11_82L$5y6b@`rrl$-}Hgu3A+*tK;oPyzXz$*SZgl;rT zXm$9JIV`Xds!{ZD_@7vLY<`$e%B zr(6vbS&jSCkpciCM$C16flic({8Q-fpm*xTxOb$!TI2#xjeBoLLBBTdHy4+DzL1e( zT%Mtb?9#h9Sri`TtMkoL5mC}_tKsxU6X(iSy=MVz>r8g3-EJJ5n4pUe;JGM{`^C=0 zJer>8yy?lw`wai|;bX!#v9NAj8;U1|#=u4Uh?$Ha7FP_Iql}kCb7P>% zWnJS?%YfK5*#nW`Cp+6etHnXW#U6v}3>Gy8Z#|A3XRsNJEv1>iE0JXb>&euHy3Ljz zW3~d)fW4!M7u_F&QL)j=o(s!GXtEmnC znoy(Z4g&Qkc8IYu5u`RgcA<3XS!1+43b@AJFWc2$V=Zn#^O|?tzEPU6J9_ zwby;y=%|PljF>g!qoa&i>?lVa;h%_L#dd`6b~h)f)a$hdxK{~)|H(+tu@jy2j{mKm z>*UC3uImkz+)zwLuF)HBqEQj33iFUmLd#M_%FvU>j!kG zf3=DoD@wda1Hg8bF`Tm}z3gfZPqA&aRIe0U8MQyPGPcV5R%FSBXS0-1Z0`7Qg{Z~P ziR^@FEf~+d@VG_!RCC$2S0tc-YOO6Ac`Po00jp#BEz1?%$vcr^uHT z*&f39#1TNw-GB`!+~QRcJ5e|Uk)adHzzl6lk?qeGIHBT@w{muR|CHE;;w>w}K1E^B zbt1fl1QQ+zB)Vasnb4NU7{&R*u5BkovNx7QFgC?%Tr0W5P{8^bIQ|bJ9IxXsVrRcU zk5r2l6yALKJ1r%$PqqkeC4O`Wnv)+t}_@->IdtYQVMK4{$pxD?Q#j30g(H> z3mC_QF43np)^-Oy!z{bXYRmCjV-sdYkBu>L>z(#`W&Pyc*{xD^UxDkb3k!PGaa%{D zsg3$tsq>7n>Sya8;hR$%A#Yo`Rk1PVxKgoDBrUUMQUjzbIl5+9IZr~5-RzuM3`lzv zhzIiSzGfz=Ir=PZ>l8mFL@5afwnyekIlQjt_CQ)G4P${hv)SltQe&g**%%BhAcK$W=kbj62DUJJ8cvcbNyTgsKad)TTb1Evz-K&k0T(;r z8yaz6zowu9F6g_8V_eB=6~n-DtyO}d(a{q=8SqFqi<@i7U=QVL@~b7ByIr6=D6kuXwG|(YT%nVZV(Ir7Jw)qBs5VjUu5o6_aqiNbSnq8TQYo#l6OS(qUw4R%Z6mz&%r^z z8R4Kl7gwK-REuRiUR+(RxlZ=Ui>s|f31$Fen8rib7##{owrQ+59@-c~8N=CVO9R88 zl&H3;5;he!d4n#8dJ*KBP=*Zn)gu=mYKXC`@tac{U)+L0Yt6V)u}~x}EB2)Z2)hhi zI>^9$?=mhRV&y{d)b3!keR{Jt0IXt`r*XarFYQ=3ni+Vu<|gI z`&q0vQ4(tmF`jvBTnOnjU$()fQUkD)Qa)w@6!iB)3Na*IG3}!rtKhsvE z%SUzdVxdkIdp1wh&(l1wa^QDouc9sLP}z{c>lndMongv7hbfdY8PGYvJW|TyE5<4= zw$wNK+~k1KU>)acAo~yd5K%{&hip&>#hx*+jkYu}42Ebvo)Qraa8he>BXB*uQKr(; zyWyrBa3l-#o& z0i`!b1|I8@lfK|LXsk-WUA|!XZV+Fv(KMN8T^jiRJsKJ>Y}Uxs*oZqJ6NoIY;T4Xp z(FfBh@|)9d@_BX%_(-qb_5Fl7FHZ8gl2f<>?k79>6Cm6 z3zOCY-6vv1B%K|lI2hbBjLuZJO~`55?(&iVGIT<6Sx__=p{VxF20OcPBiYUMN^uo? zRo)b7H1+Bxsp)sqbR_(Of3~7@T=kGSb&!vWz z8>1p^0N(x11)3i84jrczY6A&^DSX$&ce-Kfa(5jfjtPm8Gr0cAxLmp_C?b73>#ZeBVVGPLC-N zCyKKuw3Xh>VN18#^;)}=W^cO^0nAItkfW`RLkNjwV?d$}CT5qlG7Rrhw;z)NYeg*_ zAj4`9$yyFlVjD^b)AV!VWE?H~!tG{wNyj1ECN|ZGZ znP~-kEe(PYEA^^=pg4oAm*0-Gk!-!}*9;;Dz}CxlLkZC;!5DYftN6j+7&XTv5Zy0q zVE#HaVBVS@yBT;$v(;D>##&07_S$T-ev*$GBu|hd0}F%HiiOe+Objy{SW$^Kd+bUL z@U2P{7;qU6Vd9^q z<66p@O;c>Ql_dQq+7if+#ER-t1kFbZ(`*fV#XxAL6l)V2&1(wHFkrf-eAQ9+E6`Wg zm11T%XSyf-P+4}rP<=2k7gGb~N{3>IjyRSO}k+gSg<2gdr8W}zQd&<)N)vTkn{(!fek zLKZS!*uebv)PQlZP;|t}EM%bSWreD~SjecUStz!{a))}b*j6jG^3Qk8O4)dZ{5a=Y z4V4kE_{f(mxs;R}F2YuG6_g$^pEv|brKn;rCe}V}v~ifOP0^Dc%u-gM!rP3cn_Oh7 z%WZHT)EizB!s4I3Kpi0)w^FoHEO6$IATaDb@f*tPe@D2)P+n5i8Y?mLkhsop*@|PKw+52YbCqcw@}=^Z21( z6Y`53xB6`l`tH;Q9mQ~Y2Q_W-g2PvPO~}V;l=8Pd$}dT6lmpJfl7}Qs4My&sZ2O#r z-!a=NibUsZkb7Be@1L zkZqU?=YV&65u0ExJcL>MbiUG9%6q(gd0zZ@o-;}q3D3haZ1x#vDSrLCK)wx}j##;s>ImbVWNYMj1VWBIzUGF#Gph^H*?5KmY1NN8-h zUfDqGO$~@Qc(F(~Y>%Xp&_*%fZei+4`h>r1pbn%1RI(Z|2zsp|Xsp;s{x@oB{*UcY zE|m=hq~|`vTl;O=TdOxDtvyWSv&9{r*t6Fw*u1trL^#!|+y~Vbhe&0lEiYX21UD|u z$s(Q>=B7AyO|Nl=9Krm45w2a`NvWl8bA3GzQE^iFIil!L`E3I+6InzLYcbap(zFeB zwXoBkYw=EdPVKbkrLSMS$JrgahD-0ncOCiAo9O*)@89$oa-rr|y%T-|yXQ!p>P;Vz zLgaj>5z!EcsNr9NWw*2);2+iCFLM>moU^;HyyA*eZ$~I}ojCF-+zp0vNFp6;3;9^j zpFcm|Db>eob);tP6ia?{!pBA4%?ZRw*YfR&h1J%?WA8YTJ8|;K`=$;Ym{^*goWMNb z<90DR%F&`A>pF^u4J%lu4~$nE_dR^B*K3!g+S1b)~5pzmsejL}3PaLgR8_zDSwn~N8#M3y^jw@`Q^7Bm^!^C{0 zK9BLX+89T%UYK|SEj}k&oS_CM_|qI(qqv|F&FU0_H%8GUsnz{Lkt*nZ2M&Hu-yDp5 zfEOrp;TMP&eD4!w8hli<(DUKolUz=+oSGmtCinCtu0&}}360yBLO}BS2SIXCBKcDC zR3bKArgRjIT5U=7WZuz5+RmUAiM!a3($#{S7FhA%iu6$N{t5s3(R;@<(|(Mr7UX9JG{(J-FC|^baW-iU8IPNk!N-3)u zv=b4~Mn}D@ikcRuGHHJ1{DWP1kR2gbX@r8XkqX?)FH#=wik~lX!V4Uvca(R!`iS)_#Ik2mws=g2=;_1jBz zbdf#!xYByNgQF>6$1pKdX%4jixsaokR&hkmc9)4J>5P`J4=C*ENm9kFy-O%=4dR*^ zD*t)}ka#0CyrJtU&CwDBR>Sj`p3@9kz81X=OlswL-yJ-gx=?7Mrr)*$Qmw3e>3D+? zNVP=|hRhxzv+g7Xb#w~Im&2k@%IE0gp}5;7UZFK!sCGE%{+HwGzhyty$iuDKnrJ`+ z!Io>~u$~ZlBZt(ZtT)EVO5!xG@~}y(Ug^-0bnXp#c5;;2G~ZMfGc85t^DSu3FyaIw zsyA0NBAhc+OR{-!S~qN=wTgM8%I%U5fjfz)InD&D)Eoj*>3IA(ob?U08+j)%K-Qnq z#1~3g^nu*Ci~f8wPfC(7)n0A*EbiGQZMfVXhf=8=;tOe3iZ2JY_~*V=1kA)oqy-x~V(D4}|TJYi^?RGI@e9V6{J&7#W~>g3T>_|t$Wr3N3K?Ry;{Y;3y0Ut zUn^_yaGDS1*TyqaVGLeTLCD(@58NGZ4+mG{WHk~-UwkjVb;#9lai;t)?=*zZF)wvK+tCuv^;TEV zmY)QqAm1_iXN17V+D_~je`V4EnN7ZjEb=wiqV&p{8wZvqvNODEin&StHV3Jaf<{-DkgC;YJRIt?EN$u&f_?HLPHR4LOjMtgsBA zoU(lIIZE~qx#e673w5LxVp#o^Jg$J2rX0@A*fcmasX_*K2mAVL!ytU$RRHvRBa}2a zzYiunqEZaq~`=lG==9x zyKMAIQx;P0l`hAPc8c`lxY-`@5=b<6Jy%#Re~d6ZRQ^%?>&Hp+w;H`Ka}`4=J#i9x zMJXHzuy+?brT;h9zRc%;2#G2f%nEeFG5 zeiflkw;qV!@k6#Oyp^@NNvWHlj|=zljWE0Im6EN47fp0K4N+--?8b6E1tX(8u7YFwDi)A@X$p60rFVA1bsJK;s(beI z$_?{uRn3TZCiG>@Ac;jGzaNXR$42MFpN_PbJRd%%5i1A7^WnW@_NoO#pIB1tAFQjo z10kC6d(8|60K^%tnc-V-#v6=A9 zf4vS*QIpl{Z?M?)W~jdB#`$kj3QiNu5a;jZHEXN0xQO5!cs=vWGTv0&u8DnHu;w|$ z2SyFx=Gq)?w#>GCpMy@$vx%$&T}M2MQNaqAw1Zq!koij&vS6_tF`N(CECV}w%*%*y z7>*tn$85+<9vfuDhK3ld9;D;XBI0eR3(MprL`Ykr`-M%{&qmtpO<~f-&G3I8}`{={vA8A z;7*%CyZBXdKw>$YA-vRpx^hfozx?}GvBm13T-V}6mofIfueU@`4LPu%J1~|!g%}L2 zHIi);$8U_zOA@Vlq5_T8?qN_Pl_kjwzpJEH18Jz;`^g!YiReIDL&09+r2DM& zyx|C>#vaUOw4ezd@K4a=-$eKc3bTviU1gN~1v_mnrx8iUcD=H(cQ(>@vVgw0Kg-ZBwCv z3G|C;5$F@|a-hZ_(58A*s!RH+Mp6DfL5$xSm>92o(9xq!jCW)#e0Nd9#5ts5c_ine zhAllr8%>%^*n>h56Y4W*5$e^SbD+&2)OEsA<9nqf#jIy)T; zZYd*4p%(ilwTKG##xp%xyAWTdo04lP-s-jMP$7^ByEPj>t@W)UH{G&B+CEMdc_$qJ z2~T~{yr<)=T7@OQDuR@uH51FJ@t21%cFfFs=-)uI+|Gv_W3?IJsHI5doMp{&B(hG4 z1utA!HUbe+X-k34WZRb(+4lMlR2bIS4n#obD>>6>(g8c-rDaid1?p&fXnmWg2;`#k zs!|tp`;TP-g`n~yf-jCF;7D3*(}6<4!(63&Y>SSE1-dbvmuI>koiHBFegG#G#D!uW zVpAJr!g;ku#(Xr62z|u)(wj;?BVV-&rN-)bz0odJ`BtzNe9w)xdX2MCjY_>;Db0^J zR_Pm6$NPn5z3KrmbM1P)vW(k`aS#=6?=6PcntAyZzjbdhlYtA1Q$rlgc=&7jI2c>= zCdUtquAG`YGdenrgwE6aYl?n}@fFs*QP6ed^4akxXWxv=qwLJDLk9-WLJQ<^C3|)+ zyL%w)cz~VicD>Vd%4LRwQ4m~q$UCJae0mHTY2v_{5zs}fq|$F#p^bU%Yk;#TflW*$ z@Y+`@l&^!==C9l-?WXsH?L>!?Dz9A+FGx{D0qJC4_Ctl7@`q?dxRF8u*e_2tvkIz> z@ziZJnS^X|DmQf|L(G!n2}*KDToDBd)A`2$M-;gO(D{cI8rDJQ7oJfV`Ju3#=ulFn z^EJmixKy|OMu{c$TcIRSk3iyA6m0#H_{DPyppB9EeQcsk6|PM7o#m0 z5BzNo1`@b{?4{(lr?8{_V^0VhS+){`TdQhSc=ZLqnT7(*?{vnm+}Sy^cOB-`Z1&ncMb zdslLMrR5uZfk&YT!W?@Gs1enD1E}F$ikR!**iU>$(RD}IPIM?6SHs5o`PQz8E4g0P z%Vtx`jwqG zNsHz|u3}iaL3s319pI-U1RP~y9?pZ!7#^c9mYQPqNZDs`vg>?34Blu(wi{a%zl;*r z9%OrMI763(welrS88+`{bkn5;Rv#3DaV)h_Y0q++jRN^kh0H4V*<2f;U1jv%p>^P) zwWbuO>9E9OMRS(ZcSQP(UrJDP$FFMA%favlXf)w;5I%4L)??6Z_n`6E9&{zOL3eF5 zOOW){r=Y?%o5A#Gy56{eb!=xld%51?TLp1oKX>$?RZ*`PID=*+t~5+aoA62HW+@ z)r(dsRzMpeUuwec7d99ENMz*6y704Krr1q_#JaFQa;x5h{>nSRP8NM*h`mJrO_t~r z#_WABTln$5Y>EGRt1IwX1^i-KDqz#sDVk`24HfVgCk+FG6uTB^aYY>b`gmGz;t14Y zouWzjOe11AaOo7EfkgfusEJYyXQC*1M6pw7k9$Xa+W0csf;Ho*1A1`ES2!2YK}slW zsRy!Z(44Lb{bLn?g@dfCbdoswIKCbl)(11+vQelFdE9tg;N;(l2%AeyILf?bqqsNW z*l=4eIJ6>KmqGT|5xgWTp`9O93@;JJN@zzHdl;}gUJd|EzlPeT(LvVaWL@?G=9*9h z3=q_#OEB=8QyaJg(Ms`^ucU5FS&*w;BXE0NT(mE>@w);7C7a2GO7bjZkX0*XT`kz7 z)SpV)=03Kh3wAAct>{6$Kph}@FlMDZ$Xfa-LDje7Uz|}Z*V4CY`F)(L7?ah3!!>NZ z^X@oFGlk65ym@i6hhM_6Tye}nCCw9QHKoC`&f1zkzf?dnOaBIPt?jX4axFN{T2{xo z1git@VReL;e3~oT?ut}YhfiJVZWOZEidjgYdEX0}y`NG_otUC)Au}XmsK67PXr#lb z553{ip|UaQ-|Q@8sL&n-ppwo-NNy;k&FDmO7=O1>@a{;>EuH6{(!gIMHyBeBSj0t8E&Xc2wwwaNDoK8MO5#J_&D$^9Cc-^Eo7ZSuW`#qoaR zSmSg2DyNcZsmfFqI6jKFAS_YCT)npSNeM=$S|eukxV%B{!w%Ja0RVa!iP3AVuRB+*boTtQKEyP1PzI> zrs#e%kbFGSUb37#ph+tS!*X)1q=W&le&!mw!zCqKl!9e|w^h+sioNl~7XDvO4Jf)S zt&5cm0Nw9hz-ohhGdNL#z~Ak`zna?M#|LD0HI87=nht=qZtLX^G#H6=gll$w)S8{K z=Dz^k*(U(_;hKytFPW2agUTCeBU5Tkp^4|(f56b=G-BU#?UCVKax+7t!%&<5AC7c{ ziBUM|lC=yMpz;zPS`@d`Gh`nHc7`*tI&z! zKr1zQ{45%!Ef8o^O4ovrP2MtrMpdWTGsB*BusG_y!FHobs(Erh&>&z+a7SI<$n6pE z4I%7$1$Mk{e@{JWoyQW0lj7)lWs`eXq}^nFdHhL5S2-xwmpjRnwuxoH*hS?KbFNo5 z0~|>Wl{>vI(K4WMy^RNJGvrb-GJm%R{pcQh(UuZMobsCW!U2Pcp z-tiJV`QIayH&%7g2^alK{&Kyt@%BF>?Z$HyIrd-t6NMl-C^p7!cEhZ?3wjiA0H-xc zuZ52>2ogOPTYtBuf5D)@i=}8?_OzShO*1>OArb{hf{~WrQN>Ig@%kUAq9KdgU!^5# zFKxS7F4r^>4N;pk7D!>(83~l7Rfpx#Pb{ajfZf$cO+)1RDh@vVr--l^LU^qlhZH^i zz{cJ+V}q!061bNSC^X3du)u95h*jeO|8xv8S(uCtG#zYXrWAj~*2&@a30p@Inm;09 zqR-gAhCM1AGJAUUFqueeT1-b5vCv>*3uW zzxe*l6+kWiXvIz;wRiP_W<%5TRUAxkjCjGYRtPC&P)5?LApi46z))YXJ$eYCRPV*UZ6%k~7wX`UIq|6ZPar z*O6WMh4gYytXrWhW75y36vcT+An(BhipnMFtp>`tib1UTgrxA+T9j^Sq5bLCM*{j| z5#HJ8WO+Q&Uc9m(Kh_?NSUDJ;EZ3DmUPa)^eM7RiYWG@qL z;86CZbfimPI9&7akRrm8T-S7L+9)c^gpLw@_>nWsD^#h(^S>KmUe^)Ct};s4H}_fM&T;c_B$jn^Un zx{M+Jhd0b-dU^^{Xv~HCeqkBupWhX2f&nTy#+IxF4QA8|LRSm0N%^LAMsn$e)Bu~5 zDXsYqjI&4MUi^#GqjJr6N{i$qS20xNgHQQw&ptCl`J;#B{X*(afN&dbczHP2Uh81Vf;dI3`_G{BW)%trPCS%azLzPvEY?Iat$CCNh7aZ*6lF_XHtV;tEPrRQeZrEj8XG{uce@YE@BwrlCvFOhpqS{ zB9fJmIx)e#jsZ=fB?UPnw}T)Ef^b^`D`pjDQ z*svY{gx~bXC;$QXK$cpx%WiG8)2$>nP_&ah%#og!-NWbT*jN%nMT=rzUhUdbq(K3> zY;X};)KIc}5NPuY+rsmC zI4^d;u(AJ2de|S=0GGdYGn-HA#{O_Wnp5`|^jCfW91apv^p~mO@Qzs3Z@`;S&;@9} zj6Vfl9c+Fv#20 zx;7+jd-_9`Jq-cxzRHX=XHMMqYX)hOGKh`+r+ThvWwR0cv;Z~P9lfkg_AidXD$4Sz z;2_vOl~utDQ_4ou#MC@MK7}t?r%ypI)L(m_!gHh+hRSbCK$G+-yf~|{aGa~?bx|;h z2*UzGfO&5ij?ZeNgY+C^zAA4*xb=Px=_KXJ2KW&Muc&BT_Qm)M^w01&#A{ZH@_3&O ziEFW&;;n$8<&$EI3I)vO?@XXy-;3Xu-fa}6UWj|CSD}SXR&Q}afoR>fB z7=^7lQ@In6G+x^G49YeY8kj)em==LP@rw@B7zFBh&)^3IG5*8A#CYX%jvj4dT#8!& z36DBa)+Ox0Ce#n6MW|PYZk3CAQk7kjWZfwk3X=6qjjQ}qf^$15NTU01u8Ty5^NVBQzK#M(>-@neeTu+b#%;k3+_s^ADr8ar8C{go5 zsR@_aQoZV*uQ!zsBcYVSRryZ^yZmtiy5yX`^P3emKFw7OZSseh)5F+$9QrNQka3z$ z17Sj^>x(HMIFC6W$Cz5}PH~m;R`7%$by=POIe?rPs59^tZVE4QGf-v;m7tzn3t&mVeG~<||aa)XdLUkY2l4 zNB-<~iAvJxp?phkXgrn(Q!zi2ixM41&3@5QA#@b#Q0fbv8ehRl8yeW#Ip?o<;Wf*hd^2Bb`%uTc5v7nP zg`G>|NN9^EmsXYd%dgcti%UpCk#E*$&TQr}bI5Jv*S1qll9PfY#|I|K9hxN13`P>8 zrY1?O7ASwRi??ac?=nsoFcUU~Q9`GBWs=_(u873}UE0k%Puqs)eay0I3bxG_68>=y z?NEul<^ak*M@|90rZTj9U#spkwQ7P|m4T^sr>2(1+9O%qrg8xBuSJmHqbH8o!o|zWxRj8rZCU4{#8t614}|dC%#ZLZO_iZFLfR&*|Qe zE8Fv3VLQ>G40;)sq>6DzPEI+?UlGY|R*=}|G`7o6O^!{S@%ThTVF?@6HK-Dz1>92| z*pav?+3VCwoahh+XS-RN@6h>qB>yZHVXo8t3nN11*7a=3ye=(~yogRv7M-$OjD znN(_#=vkQiGD~x(=lymb_mfj@FvBady2cOiZf+7+hH|O7DQ`tPvDQ~vLC@OyH$cg# ziZlhTZ%^-$tM^F%lm^x?^0)Hxn2a*XVa5|Rrt)WqvxdrlHV}i61O9l<(ZB7R+HdWS zt%m8?2>UjM&D??m)lpzDYX3GZ)XtxDfTr0mxFa;JFvH~F@5!;@lvDm5eJar?=yD8o z9YxfxoPLfX-uiemU!**KUAI1lW?An73UTFKQ%wJz_!q~P(%IC!pir}&s~8;jBE^#q z%XE7zPTAGDsj<_AV!hoOTb-+oAPe1?E@Dd`iC@-!q$@|Bb4z@(6ESk&CsCCdYfMTm zb;hhulj~J9He;R;zrpw(!EdZcjj8p{1W=@0C6-gzW;Pv(^q4&Pwl&h_aC!3GG-iTC zzv2OUOaf$Mb5O)fxYKIRM$em5Lr=_|krIt+#?#Rux5qJ*1UiBH8tV;O2hu%(aUwa(_`nqlI97v#vut<^Mw-8HeC_OosmVz%yrN-x zO&H^}J|sWcVohG}Agx_53u}AX_Zz&1#GToGv*6=0ZfacIHB8caTknw?G8|V5f>4Oy z-H>dR6SP|p1h|sU$pvj%S0a5T%l0`f+j77x+t-8$H{ex|5`ar0kpeEUpbgx+Qv>cn z;=}bai5gU}p=KONOcpVNT3WQ{Fw^3QS7x z^y$p$b2KU9LHLE9iT5N>Yk0w>aL>G|uOoRbX&(x|JX?J$h#qGc^lh)c@RDUYbZMs% zT1}9%aRMRdt%+u-wXB^^#(LSyO8y<7hJva5{lvsWBOS z`9KOl_{+xAFGp%8PeCu=a+_R9%C30|y3T;s7`OE}ez@049LW*u2lk*}O>NNIXlgO2 zq527;ZyZBUo&XG7=?Oqr3vQJgAh_U)cf50!QapvrZv-K?v@^m1!}aD*&rQk=vo zofGGmLWjN5txBy*s`Ot$VY7VgH0#l;@h@(Akn6@D)jYPDs~7?o9OYO;^4AWZZsf7K z4^OJ(9B;x?&q5`L!%QrZ$wkG@V6>0dTEeYlVvCCG4oM{rc*GtbS;1|%kBxNA89ge4 z`OT+bpk0Io_v*1PP$UKVE3B~33h|Ce-^qghF--(Ha2E8Kg_11iSSa-~W|g6%zrr?v zlc@nXMf!X&Hc&$e)=}BP8Cpm5TD}?7ArTP&IGL(#nyW=MLm|p1VEK1Mzdo)?@v_ zqA(uxSKa|xvM?J%)cDucf-F!D6G6_=x%xtmp3oefqxDZtMqFsSf$Q`)uXj0?_4M11 zV|nLEOp#Aej38!Cv3QjemFty4X!&c2lsJ`}z{S!i_@A1NFK`uuLC3ns!1vxsyMSpY zy-GYfavbe;*U}u_U6YwNtK2h}kterc211@WrfgN zeEOaVUk%h`p_$!KG%Y@Eh{$7Hq2!gSjS7VpJ68Azloo++QZt8<%23b(`rq_hYNFCe(OP5nMi{!GkaoiBoOCiGb{mO<(8cb&ndARYu2nTocktioN=6Goc^pC0Qx%+%zP& zQe^k66mRE)B#|&#$rJ{H;>)=J*FbFxML6Uw30w5B4dO~_Al?||Wdp44cP>z3X()r! zLFp|lBu4Y$16P|GJ@>6gs3a2CSJ_7O#ne!}Y21qY!Y>Et<-gjhkA($CQ_u_&8L70T zaFf!|hRPoq1i$6K1a`2i-%DU!VbW5f-ntXx_hcSg7on2~Vd#x{{}^C8~QDP#7+6xsMmHoAEEsUX=^j+cZyJ z!Bw;=8r(t&>V({jRkREp2-j{s-|NxbBu2A?{uwL6cv+$7rrYgtg=>dnTl5qHUcv@W zd8Yf(3FFc1)KzdC-{l>vHtN{O$Cm!NJgiFG9NNeimh(6Vn&C^x#W^kVx1m^gYHJK$ zGQXG!pVd0m24xxYY7O}a9*jn%-maA9#~Z8kjb1k@ZP6qsH-}3_E6b&JYaBo7?R$%| zgTpI+>)uDga;g>iwor|}8OIT@Wthg{SVY!?r|N}#yWSl09^?FKC)!Q%1$MI>A#O#@ z0D98S82>z4K=4H|-_AGch;?mcguOnQah@Hl#h)GI*9a~=1dNOq=1$?WUh`YoEWN-A zZRHGlws&nZPoQ*mdVF%CHZlqi^-1q7-YiNzj#Ued(Jw9fwR8DO2V{A)wD@-ZHs;Y+ zq+_TzTcR_ZV#`++$5A@_#7L&m1SnZmd?b_Mg%1iLvxHTlR6CaqI~}2(x&PI$0i)6g zw;-w-hbe2>X~IYbQyJb(5fqK_YJSD+dq&>|MP()@Co>tEEa@?&mOyihe%2^8axmlJ zPdR%ue#G#yvluCjEw?(tCF#4DJLKWtsH~MSU>M1))co^e@cCvlznVRDYVyq3sp&Ik z@M>PYGbu)J5ykNGOlGyw^oylJ8%=QWaRi2DsaG5%4PQ-ny_*y01~n?}apLtXH8$qS z8e`txW^QlK*VA}EZ@))fZ){()zE4!juasI@oUa7)j7^StQwPQ;0T1=IQpsaYnQiVH zUKtJxuc}9@@=?3guJ|bQc#ktGVNHt5$%?@5rQMg6$7SWw7!S40W#w^MdGx0|l=;Kv z?%fHNSi*7my~=cw9tYcaia!oNqTD}shV4X+!a~UEkq}t*fAcS<9)`Ya@(Zt%aQLpjYcbP>(9IYQ6Z52wewQ`)sV|0{_W{Sbsn_Smi zUiDEL%d01us0;t92c3z>dnlh zHgpZ+QCME?Nh7pAVaW-j)u=X-X)P9o!D@~?3;S}8jmJaY0oXkaAEs}6_0SAV?J=?e zCl5`I!zdM%#}`U4yvA0M`vW;qvm>&f@U1%DHixAR_cmUr`wMf*-nmRjgCbv)gtWCl zGeM^$%sq8fw{^X(o?_>oBj=Tps)X%qbnYPyWsgTIxL`bwyLj`6STv>xdDtPxQ$)_v z&pPgaY9>rBk(v=Vk5;1nxOr~tvLdMe@~TyPUTw3K(cG9>x!|5AXPYO>L;I8&4!G(s8ZTfXmJy zf1-g1V*b#d_~y;ppBOAw$3(_Lzv&em$Ql+V4$=S^&Vom}4OHDS6=i9?=6 z-eOacp#x}2r4D6;C>fG+zrJf*-ps5kx8{}Il^bYahE^T0BGkwoK(?^ryX>Ey+Eng1 z_RtM_JWS4eB0VH)+ZPWiy2;_OwmlYV+Y~UZI4(%BQ3pDcwXy-MW>WiL=aAxRhNeTr)d7oR)OIlviA$5vTDT+$PyX9k!=6 zDI1g8E2>MWqxK}WvQ%VgsTXbiEG-W{8^)JQ*C=g0%0a2_STl%~UFac0n58smHqwMb zd-*g;;86KX6EH8M5MKT|h1ln~$~qF@&eH8-m-q57*oQ8UJ_JjRT@`@y#EDL8>c8%B+td$Dnjz5D&sqXG+L`}N$&N3@3a z4D0=Nv@F8_tUY-7Fx-7NHs9lR-u1gL{I)wsu`?$*H{dVJv|z+b6rXchrW3+(F3a>f zLTP%*v@#FaDyt>f1_^r;%Vc42NTJ+^wypfwy@^kJRM{I#VLMTW*r5C3WVhJkF$}|l z;p$H9N5C%2NxMuLi)5DxYcU5OIySoLP9(!{l&z@i6M9>BY)u+|`|i+GZF2QvSnhut zr>~boR;FZl8PFIK%U4Tnq<6w?@6wVSsE*=|^1mlU4wc`Je^Hu5J>0ab@ye$ZDPQC& z<#T(rQx}|;c>T=mLAansM<0&{hGidrWQ2m@i!HB$NXv@wVCHGtf-ltyuOPwsB<}>@ z5k($KxaAOr`cj zDqohJi*VdKf?IOv%oj|br=KF8{LXXK@j`GZid6 zS_Rl)Em5GFudSMW=f~i61rT%$02}St&Z*-BIhGW-@d~_$;su?HB^rV7eM%t2GT)As zxPQIqmfW@e1{A9W>u&=L`a==;Z1odiRT;VQYnodne%R0OPyfn!NYmtBJ0`$ck~8|> z-7?tf!y#KH=uUYHlTq_~RL$5+sb)_Lb@>RXH(YT-_Y7j=7^(qUj#K49mxE}-YAsKf?9nqQPPE5)F1p*Hgci`F zaAFjzzt$Kwjl$~%$*bCFc(_t1k84pD7I2V(PRbiCkdu1& zK>=`mk%2LmMFO7IJ8@|s4j(m}7z*@u(}y3i0L47ND&TEWs1qg-3}F@Oy$*v|V$zCA zfa%Q<3m7M(0TYTh3ObB>!?j@pXv`6h0Qg0kUx3=SN4LtZl8m?7xQSri#~vOipzU?q z({qv)H;2`LG@P2^Gi~a8u9&+(0Gft+jkgtUZi!0}Tb+4g1UWx5i2!E>c-Tt|7;iq` zE+9M4-f)7W>u?GBe6dKV@@qbVPpLMj+N>8l1p-iCU{-2lZmnS~w02TUq@I)A=OeR- zJ=r}jR}UNHdO01Y)}GS4L26B6O19k9)6MLD&1Z z!RvaMpozq(oE(gpp@%l6t|U`#v>C;V>~6bt5FX+y_IfeU^Iu4HYH-LGb{sAi*cx2x zR~t}_rPjswT~wDuZiW_>MCI>_GJ$wCRL%`To$`JvJ@VFG1iI45#!-F%-Z0l#RU(q| zkHC%y<14PLY5X$Ryk(Ga#c|E-ZCEva5l?}yzUMp4N$}P8d}q4Tpkj&H-Cx{@x}>&3 zx7yWC_Fj>{hLrsMU^cl?vtOJ1*9_kX+jQ1?99g(Za1oG^WZyMSeiE^M*EXM3lg)0f z-CY^;xwrw@YEwBh*t!#jZq!xbwe!oAw#BeCk-`gRX{qW9!<-ihSSwgJV~^2T$tPUI z!h#kNv^b!MGT|H+*zzswc6c%0AWv0gb(F5r#qGRe1Qux2 z9(+w$qVVXXf4rxenvdv5+f1DjG#!8u!^$wVkrVjUq{m#ojAbe$yz$fqL;pD*NfZ3~wq%yQ) zO>QS0$JRn-AYb0oURfO<9fgynPBR&%hB914#b*BeJhBJTR4g<+Z(DR~G;et`{`q#j zwhw-Ozr9p1ika7%wd9@v`DQQ&wh%u@BS;*Rn3CCN1SnednrIs*giBR;-RwTlINUkX zg*ETPF;&9YF=XzujVmvGoT^bs;7Od@rkS0t$}BBlf1=jLc5zE&>^xrs`ow03Ylda$ zF@LgAA-s#^&SLB?!0o8}67Iy%?Ieq!EPP*TJg^_cu6nqi4NqnT+v>1R9Glm)lj9XS zOPh9qiyYQHPl!BCx8WC1hc*8vQds^_q!LHGF zV(`umC`G1)M|WvaA}LQ;7r3$sO~YLlI#{VSnfizt7b&J#eZi zx(D=ywceiWxQgHO0#jqN>`Xkc!3Ibv1HU9&DNO7`PLws+2IM%AtW%l(7u@eX=v>Z zxg20_V25+ud!l}xcJ<6r=e@;x9VsBKQNxm0LQo#rD$ROlQ8*st78=1pCYyj4$rZ7q z7vc`vvan$X!l3YCAgbar3BGV7P%}jN7f95G%AXv>@L8f<2EVJVp8qzOxEeLJxW?*- za%V5^H5hSX6lN=q`s*sd=TN-vJ{9AzF$2c-r;X|*^oO!AlwWNxdXurV;>p{wow_M0an zXpZ#)@P6|jX~B^A=P;KW%*|$FLmW37>FBc--&7&%DcY*B2DkR&SUF?#E}&EfZ=@ei z4KPN|kdxEZ+bOh&JMUY#P+=+IaUhIAlX*SRg@gqBfEI}^j1xBCc|whQ{PUv53B zy@jONrL~~WTlFrWkXAlNY&cYIC*Vz<%m3`~9WtHCJGqK6<2-wk!`v~4beaep_4Dwf ztLa1zj%SN`KO;PyWNpCrtAitlOT!ACI+N4>0a`W`&_LCIgck@G#-#)<&0g zv{8;Y+6nCrpePUNETAad)o?Q-zf49T47-eyL|A(yc@*SJ>Vu3vGsc$;yfDId+zTU0 zu&F|U(Jr$@7zoEOUJ@k3N9LAsyAUmSGQ-P<Kl_uR&(ETocz?&gIa>)GMR%`z!wL$ks$1y|d6IFr% z_F(lAPy$V(P(t5rlx)A>HWb#++E)v8rrQtX6W0J8H+DHhL5XLu%~p~UqUd8AxO-9q z&UG7?9|6nFu&M+k*hBl){J&5nq_QLx$A~5ZQ=G~gI`DPl*lbu z)-#3M@{@$uq4F{Oi(8>cOM8!2hL3R-!vOd_7y#JXz?M+hHL|yN8o0_Gvuk+bXn1PU zI$-1tE9(mDGBtIo3-|Ew!qnv2$gr28*&NNtkjT0?!ZiH!p##HmMqWbD>R1L-K1M*t zx0J|I(4|QG+&*77*VTN+#Mm|o$C<&LK7~bOWN_gk6N8$jn|%woa0tHNBXbA5Lm9K^ z17mV|sm4F<>5C+hTw%4|k&pm_sbPAM;WJqrsup4h-;WXAbOc^K`T1RW3mko2~ z&ICdmsw1%|*zdWBbQ`BsHcd@R4MATO=knC`Ws!QxD(^PUz_Qap{V=kg!w++h$p*dI zSpw8=aU1AXYCv~QeO;jK4tBc*PZOeNiFmqhhU7{-@OK-?uTKrg8zi%{dN9TvdEpwa zV`|eVGOWC#DJUIdWT`Z_et7g${Y=VwP!c@#c5UYQ{?wS~w%&@{pjTWe7a25y!TD1X zDg50AGZp|QW^w8R? zsS$T-;OX4}Y8{nbVE4y18aX?=?F)%3*DG5t{w_7txmE*O#rL}!FsuqPi=)1=;KxtG z55;?(@H*9yp#{N1`33@Q5fV*_4lsQvM>+YF#L@oHH>Uj1nWzoOK}Lq3nwl~F4T#p} zqy~7b3_e_0+Mr8X;p3JU$Wl}{n}EvGBX{8`z>TRwHy1pb3J?b8rPLs}amNGp(%0qM za+G;29fTSnrPboVFd4?k_<;0Pq7@8U(=kfD1n%X6ze`)-T2|~Wy^l&4e7K>PN!;N( zZQjg~=!o}3=5Jg(3sDSr3cb^i0_dc*h+99?)tXupg&P8T^YDvq+=}$Pnwo|b#W?}` z-YVGpZOUb;Nvt^|xDL0v$sCi$wj@24zJu;M?INm5Ul%F9* zGE{y#0Ta?4GW|0O(6@3G?G6dB&xey72&RlCY^1^PB%n``%9(GVI(pP+#HdfvKVzpb zel<)XdiUAm7xR`ai$$4Yzr25HjT{1eDM*>}*(~O()%;vJNrJpC35ssNXC(ZuO=k*m z%=M)NTKCQMNB-1ehNuxRn}vT&VZkGXL6IZQsAL!*W7My| zf*ceT9Nxl6pn0iqpE84z?d<`$|n%h)d<6Dw^8{c*FYAKQtYt8U&l*VPueiQCS)`z zlj^rEHs6^V65p7}(FQnu6n23Zi%7$EIvld;`enx$YYL5R1eQnAkYWRzblTepL~q`Z zN>Th9Y^)sz1lqRgW%U{wge(OyIp}J^4uGaMiE?bO=#3(JsqKjHWhBa6nm}) zLHi<^+)AGRlj@ZM?|i3_SN>fZ{!sb1@h{F;lqS+UwCel~uCksxQ@#dMCkBkTahg#Y zt2BYs8lJi@*}DtZVDQpc9K!K3Ps3C>;CbF%?&yz974H#|BBsZ?)ZA&UlWTL~Oh=a+ zct#=dIFl)(k;x3tfixx3JcgK;ML&zFI;^{YYUChaqd}!gxnMlaMO+Eu+yZeAy9F>lH4S@5mb)w0{PXZO zAn7)aA7FEd#;gjJz5T#L$Rx9uDe}6Zh!RaUXa6QaE)(u6_)C)R_1;%7|0ZQ7{dw3< zR7DJWUjgY1dpvrHHTw~$g_A0VOlwHCb9)7naY9IJ0}FkE%mKJ}hm%55Dxgxc)fRb{ z!|aeSp^Y#$ao(Wl7bvkocFMm5>|9Y5pE(CLDNT;f-qMql~T%!qu)XWm>quFd$Z@&hOtvk zg^X|n8CItjrdS3Koh`n$XKQA>%*m!8K50h3>yYJ9{v{i^eIrdY^ zLlvTmHt3IRoM47cl16Ko%&tEuxlDL`UH(HynMcy#90Qtk+I#G?7agHwI-tclDrGcvlP6AniT*)1F5j zh=@e;BDr}+X@4mzyhtAHUj-Qe4N}Vmu4Ba~M`Z^5`-Nh>^0yGOhsqc6FKW$2u8-kk#22(0_-3wR7;G(M8#(OpZ5r?{ z;-(F;&VT^|H=j8$wpm)pc?he5Rv~adVrfD;8!4n7&OC!u?FcMvl|m0G8X#Atu>@M^ zfa8>flgBMX)QGrGSQtzhIuf8gnNR~{pNSI(HQXBHGLzTTdE^- zBN;!0Q*Hy<{EgfS!T<|z8v$sTS?Q4+UIxQ|m*0d*ipxE45fa*}wCZ6pcyhxs+_>ck z4>?1VNai5&LD6j;ot7{^K~>JKU*W8g>V~F?#R{CsVKbeoF)&mpaN0@gFyo{1j@*u> zleV5b7D{`?A~zJ}xWGW#bU6sYAmR!McpRQWLW>&i5~1gUb}Hv*o051fV>QcZw$K`^ zVDtoAvzcM+0BV570<$+f0umVwP%ZLfq2Ms@NDQM=zMWaj(;WhM3MVpk z$*pMP^JogpXDEqOnuBP)6j5g`^V3o0ku=1>04JUH28PG5Iv+_xCQXP^X>UUirJ*29 zh*D{9BM=?gxVAFjNu#xmKqD3=8&jMHIH@$Z5qKWjm=ZTY*=S1}f#UFn=Mn>u6dKzI zECyb{>$crJrg{ZK(5r-_S8bK|GUg4?$g1tyzu8U(Y_3?(o%gYOQ*RhyEjLDN z+lRYjTeQvGj}cJPPmRZMJkk0zD#MF!`0Y=ZxZ$* za=0)XQHX-PdPt-!uhQx=H1RVcX4JD724`ouVB0hmH-K$izZMc7c z@f)l>P&KgLIAPof($O&s!L<@Al8E|&q8Ll&mqgk~mdtO^gp&hc$=qh+-+(Ln#x-<@ ztjgAn@Pxw~CaN*46dD^C62m~99GJZn<$}TMI!2nTsSHw+j0#Pq<#hIg*ffcp-Xakl z_GvS7Y-((J?A|ePwvB{So_nXX#e6xX2=t-FF4t<$O=5!qr{x~kFHX>TRd2sn3J=hG!&Q!qB7LE`#&Xr7#$NW+4h4Npco}5Z&*N5o=BVY4|hmrA+c5~`t8IH|mdjO%EfcJ4Y zV>H&a<5=KBqwtlU;R!wbaHu#hJ82I9T~DB20-3T`VK*ti%FE(WQdqczt!*5%yU9T+ zLmZs1!8Y>|4Pj`Ro!x|%`o{H2@h@whTuy?ehq9$Q?ys;7R&8Z>a8gZW7^Z=_z@x#BeDu;`2 z_V7NG+VJ+Wr4p-hbl*WgrQ?CxX6r8rVf3*L?#EIC_hzTQG$8CzAReeuX}l!sTM6(W zRcvJZd`igB>!A~uU<#0MUXr$Lg#3C+2-zLzueSX?!4|Ww^T`{oUkpPjdDk zz^_&u_l{$Yg^&@EYRL)(p|nRCz7xy5fIM{SBMAX+7$H+A3}nGjMKBDVQNtNA+`bX& zlJh>FD&xWnr5ZLHZMLh_k%Jj>d&V7~CQ!mI4tP4E`@z&V@$bHqDdoI%YaGcx_ zshw;jeX}-_WY=sYT}2x!21MewYk-cWm~6>dianYN!x#ffslkCEGL~eIroz4^OzAYZ z5tz2q(GWwJjL~(ya`mIogFIn0Rc>1s70yLKG4JO$t7lx*CHy%3L zPFC@T)M^#Kp77?A*cNw&Hy8Z{UR2Hrzgf0D)}9<|%aG!x&p9pIlM`92WT!@6yKv|O zh_>#Ihi*?|tuZw|nQ2q}IA;XH~A}X<*qGCB9)D4xtK7o+(Y|o4TNg@4fxypJbzdr^6@A8iE=^k35&>a$W zoJ7U-hRcWzYGwHNE2SX|L-r97*FXr;JDa!(pjofVB}q7eiUkVZ`+hOgJ*mn^Xb%_l zwJW9h@y05Bqa#I)O4wv;nQti^uas(tL7T(9zJ5jaj@)|3{Y5O1TJ>gz&k^D5QDy}* z-t?$_iYHqg!=${5H}~zocXVaH{kA4vXJ+vpZwhm#@QHLCtt`Jl(Co??gmqyynZq!e z{-P*{@A=B&c&TWG{yp201MDC0Lh$>6xXXx0QN zi+dX9=o~M)I{>dau0zgO+N+0VFf`l})Xq;Hnj8mMqVo7cshwRJTg9m*Tv?nQksa*~ z+hzi_3Z>;zTMZ(ti8;-07W_tg4#eV-j2G(u!W_&P0k(kGI2BdPP7^{g2E4)`GBe`5 zF(7Q=3CIAGgO6kJ__1|}$~cW>6bnyXNfDuH)8ms9)lq@VY<3pNWGs#3fJ{1B#lO59 zHZ0otQB_gmS-fR1>p=NpQGqF1$z)b2a*;`ld=vTA?5R_eXU0xVpE)yJyMe zeY`xAS#98`cBw#i1eG2~&cSThvV1k&^)6%>!lX0`Q04#(Guj3rtNoQ##6^{9Z`4~Uv4CZ>=umS-Nw5X6hTz0*46H9g7gj)o?D?!RM=xvq z%NqZ(#y95s%NqZ(#`nm4_!p<~)zX3&h~Jg4u1HuQemkvB)_XSjrB~Ei9>j&$2w%g&njMa`v^p@@aqWTx7R_$B1x2o>tLeO}k=n@zYvGV(wNR##oJi7M&%dNcj ze);bdYUPl4_u|mTpw?sDU_*dQj!W)^=#_l0fKLM`NoVWy)}2VY1B0=e1Mw*0GhBVE z93F-fKo;$}Q15aC=!La(1n2<=H&sGL$zkrgO??rjNqLCQmLDCd{%IdhR#>Y1gH zr+4mSFUHW)W@6a#9eDrbl~MjT$s1hV!eRCTl7j4~Hm(e|g#lo5;q#f`dijJ_$ropW zYjI@@Z8Ekl20M!NLI*d$w#KnC!a5I|Si#kA=My~_-*<83D=yL*xa~8+)N5-gn z6ZN;^Xe-X_AR!5UTqnArmCQ1JX2PGHW(B`)0J`Vq;UA#4=xz&cwQa6m40bNm>uriy zGhRMW6)Jh$ zJxQ-`2rH-u*CRzI(viS-i*@hCCpQI1-(FEaw*))UNEbV|e5^d-jtFZqdlcGp=Sr=T;LYuGxQUd4iUCMC%4?(~EbR}j2QX0G zIM_(C2zCjJ%`KEF01k&ew>9AC=hSNenBl#ip3FnP&k<7x-cq~WXdRrGIDh^;TvF|& zdTqSkT%0KS=O$3HHE}VxnQBmrRAa79`O_dV7lUi@Ja_lxeL%~$=0YLT4L#A_+%CIm z*Xt0^lHXq7ZkF2B3cA?Z&d-DQi#Ue-AqOyLmLF8gA+& z$icUSqvFM{3$APC&#Rx1xEoumxPoD2ZV8d=>^Z#{Tve$PhhL9Fl@O$Nb|AMc)Vzs* zi=T6RkF~(y9ni3d+ys?`Jb_}Qc;--hxzoU4wn6FeGBlL)_kv04g;H$+q?UrelT-n2 zm$J*<=vU|cB4lwD)LQcWHd?#6P^ssa=lp8ERGAB}vE}h?_G{-Z&UEI3YlRAc=z)vi zxF~_I+jH$aZia_S0nK;NHCE^enh>~wv!w;S7woj2*7CUYlpdqC1@+@<;{GB(zlI)) z?^&(-aqd#7{+aU4NOo7=NgtEf;v+{NBir$D4}ILR10T21$Bv!&*hU|lZo;JJ|=F(#~6Ja*@KUV>En;-anl>{aU*?v*u%&7(8rIy5g#wp$I>u9eERqi`uM--Gt}stvncccecV5Sj|2D!_CRLKokcp?DqTF` z)y=4IActbp^l|+ZKDOc`xK$OC?aRwlX3FoLMycoNEnI(;p3a= z<0P)1F8>?)_(u8&=;Kdug=P7V=%a?~i^|LNaTNKY%Wt8N#~;AQBlPhp`uJ7)_;)yK zQvL?|xD6pr> z`k_aVZKnJbkBUe%!{nJMKg~~w1~ZKRneq%jAuP`@hG!VNGmP08#_9}XbcV4x!@ko+bbaw5;Hx`{OBu*?kZz1_%(fKv{#-g)uGrqCt6iJd;boP)WvFJ>b zB(dl`x(DA_bpDLKvFQ9ZNfL|BH;^Q;=={a)_{O5MnFnwd0`6l{FGE;t-MCkA+J|3i>AE0lK)5pKxkB{%7 zZ_{J=_)hxC;&VU!9398U{Ul9KP2gjae*O-9J5C=YT;)&Gx6jkJ?exK-v{{ML3-pae zX%orQQ4*yq=!1po3HruD^)2*`h3YwyEEX!d@m&ZNU5+Y*iY{joLPeQ~g-}rfJ|R?; zmq7>>9qJN7McaNXRAsXJJHZX+{KqQ!hC&H;9qDwaY4Lfwl_!H3g282WMDzztUp;f7lZ2xb*xM=TVm(r zVsOpT@cH{N5tVO)06w|3_0<=Tlhu-M!a%_k)uOF9n$&hGgUWh*oqEjkA+{D^TUE2x^xS8Sj!Cl50ta z0PP__0`M00|MuhVYV{}r*+yQRgk*=AFO% zIq$lb6*gsSNAAIMB`v%~LPoz9`NA?C{jAir4=T{uyJGVY%EtxY`Z7wmx$D|FMO@bN z(_&0rk6Nq6xUk48b~@1b%SUl8cod2~?gn107aMwaydg0xjiBeW#j@Aw_mq!8UvdG9B zxX!|)_6^oRPGXgXy=GwYVZ~u7D>-B)EGNsY0FSjoy(W;nW;tevr0j;J{_?i#ahf({ z%bl{hK&>pbg;@%c70k{MOG|#TL4i}EG6b#&XB83X*&@~#79YB4eqX?36UbCwh2 zgtaZ>GT+0%8B@3pz;+S0Ha9;FhqKvifSZpiIZzt61iKkR(hR-37c_#!pg}ClP-Tm% z?t_2v&G$c`EPXwRx7-BI%FWr*ja3TQv|x5py5TlU;i7Bl82C>t_m}(b5zv&8X0V#@ ziOEBI-@E%|Gy?aW4w(bAp+*T%e1KO%Lt^Sq%tBf^CtOTQW={{=rZ?r6PLrS?%Qu!e$rSgbf9th4vtb`=QQNSG?`_+vPj0y$hCNnA7-jDcVwI*SEONlM?>w*hACUZ8 v|2q(m2NGro&o+ZSQ#ovAeuI6`u*;V9lCXyAi49kxJDFr;y*Z&?#dz=^r_)~! literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/.doctrees/environment.pickle b/doc/LectureNotes/_build/.doctrees/environment.pickle new file mode 100644 index 0000000000000000000000000000000000000000..c99e9ed5a00a6f8ded42264f63025604a2ea7a65 GIT binary patch literal 74321 zcmd^o4U`;Lb*A+-ZLBCzU8VlSJ6Lz%&VDJbt!LD z7cAGQR*Y)B>5tssu*#*Qu~%y;x1;j5FzYzygLh+5f`U`4TTWHTZ}CtQMYAxgyM?;3gsR?p zhTmv))ABSTkF`Sf&USz_{ziwYUG8?|M89;Umh>2E+uW{u{GKTveax_oa=H2lpq zy>1rjP66n6O>@E@c8!|jS#`%tP|O#dO4JHYnakkJJFx$jRdJ_knkK%N z@)Sg6#&Vc-b<_c>y`p>I7{zo{V2V+Jg;GR3fAUI!{5pcMK6^KHN&mw zWvhA~ePHdUKGyWNbCl;a+@eA63tnSx&RQad`l^izHSdf5&YE5|%3gsepiFfS6;Ue` zJ&$tq8nv2hcwS*qx}VoPhtgG+y*m0^MYqr>^`I?Vb!y=^AF9>L7OL)uUe~olRL>sT zrCu#*sw-YgIJL5o4)XZj`8w>!@?^bq=Ct{N|hUR&BXDTPT`F@qEoeyGP{=%|2bguV z-Hn8tHMfj<_ZqWa-KsZ88i2_WCJ25C9*S2%E<_4#-742nJL{>zqHA)yMe^;&l2PPu z3wpWiEEXWO9G78^VyJZ*l^VDoL9IEBdaY51@Pb$XO6xt-Sws(ulGm+@fke%jYyBbB zFA7CSY=f$3SMWyC8-+mp0D|>E9z5=fNGSkg`1za5Iz$|*Y~HG(TKa3HIer5fG+%aR zbriEmfTNb~&n)VvsXOMEAvzlV2G6KBcyMaNXRDIXb!Q&pL-oi|Zc{4_-fd)1s%TA{ zgW{8cGBExj+@)e z?dA@1r@7xeU>-Cl%v&#-x2>2jH}5jCVu?6$Brf$xfMYCiY=A6GRUgm;llri{Hm;JrGMC{Y@ zda2@6EozndlIzq;&SDjB>-v1b#IOj7heo88tc5(pP+1;h@wv;yt5z5hkX9L|Aa5>_ zqCh!w_VLfcmk&E@?T3W|1{UUPCKF%6e$_ zOX!0a(7uI0KlRK~1GK!@%Q`7HP4d5N*IMI7MP#`iiB@}0nNWgRsA+RJo!ZK=>z*N+|ZoSo^76Pnm<9Wo=m>l z8-Mjy2m_-Al?5YKfhwJcs9_e@Ew6_7l^shoBqL)mDzXr!*3Cfsn;vl?^^DRHt5^qn zdrlMVdDkP@0Y%+yuo7AlC2U>3Qi80e&Yz>s-=}KD{8+;ulCich=a5L}_O($pD9HLV zlB0GbDy0uYG89`7I%du1`PGIxjTfk$v-Hca28&L~$jz3WBE-tzaSLNYb)I4x{xFRh zw!VaVW$hJ>ZX+1nk5`=UpjE|iVw@nEd5(vRwNFd|j3@Rq4c*^Vvbhg;Pve~DZxOMLCeYv!lJYU%=?_9{G$A6uPt&|>wILtFSGA#+ zn{_A_eG7US%8t|+a z`*TaI>ZIInf7*psbs_(B5O?8e^z;i&$>2xLcS_(R=0}Cp7S$@D3OFSI3)$4D&*kpO zdDgsq!9tZGVCM`Zw`|JYTy`5Tue-$s-S)^ z%2y+lpyH6Yf+kG*go=1+-L)!czD}J+aOkXa=orvc3SNz-5GyKE(hdru(V-y-OEg0L zIo17`2{o=vs$5t}07UVr$~79LiKaY4e3-NnYFC~N_-Io8tJ=qWd^8Qs9&Nv*ADAv% zv(wV#4&?8c#+<{*f!*`a!M$lU_>#Pp2^X}0AcLm=>gUnN!i9Dhst3um}hNli#fgbF322AMUp zh9k|7o1c(WTS(v)WYifzY&;&X)Ls#C1A&ZS zA+y%jAq4x`zCqob+Wb1yEVK(Abc}}38{+Yg<|Zt0%&!Or{5BAQ58T(!)|#vQ_4J&-o)KkXN2W$idp2Iae*V0H zxj(;GjeuVk$(qyL6Mr?2ghud6@#`9)vLQ)@BiJ_3rIZ%J5#|TYp9$F><1B)yYD+p_ zY6)`@)za4qT`2{<#>7Om%`mBlnFp}zL)qe7KqeH>kJ0~v?l8_7SD+a1`@*7CqTZyw zk}|RQ#%z-G5K$ANx|9@be;`v!^I}+}5vB~|SJAcDOx0G^5)Y9Kse?HZW)7`TjPm!3 zE~XFFc^Z8%WDde6V7(5XKo1Bq;&~2_BwJXK<7OL-1uXt}7DOp{^213-I4KI|)(e<8 zV=Tn9h-1Y5CRRBSzC2Ch&uN~-fe)Dv(MX=qIb-EcuN`E0i7tFGhrXvw?qoj(EB66n0k{JJ_ zh_IqY42d41BE(!rLIyzabYZl`Y5`>o6NjP0WvCxCe!MLJQP~^5)q=;m(W5KLgTh56w zDeo;n``;Q%8ChRO$($0Z%Ni8H*7C?)lhys^tN(t(RaXUyzXnOI*9_PbjxN=vFpY)S z-J?A{sNtUr>z7;Av3 zH|^eYmo`1E-EfmuG4!gZktJl&#H2%mmPMuyMQrfG;2_c+hK4o^9UN(3?A2j3sV!?# zSPe$vBH?3{^4eooxvb3^S`8}`pg~lt7#R@I>)N6orX-rcvQg$VRc+qYXE`f>y#B5a7nYOnwXLX!|IZXJPbz9g60%`eVE!=wZ4zp7w%7S z^Pv>`P!Lou+KG0LHm8?8W1mJ%geXX=E1-hPeEeKg zLsIf%P6KNW3ImA*mfDnjO%)Ne28CxHwhya5wfl+c?mh7;;R**K`XtW}Q-X;UnMz5u z$uaToD58}WLY%9KqUaxe*U|vmzo$gfBDtXXTEjCXr^ol7i}I&riK?fU^e^!}6{PLk zkgJYVh|TLeG~%wfy<~3EXu`()&>Z?XSG1=sVm*-ivTDe*Ag9eSsE`L$&2FF(px0{9 z6|p{`m*qRANDRlSKF(*X0|V1|-Pl+i67;3DKx9$%p^u$r=?!^>LEb`n<~8&VDi`%P zeZjFx8ZBq!+(wmq8A~_FE>dYjeqtG8sl4PphVo5KJW-u!X>74nBqAbdllLGTjj5OI z({i_{zQ&}L%#NjmtZg=AXqORO3W;Tuz@W5&4N6LZ3=Q|3 zHC~-NbmC0z#EIjD2hW~3?zjl4N@N=b8ExC^D75CP=@=ORB73bAFZ@{9ZrMt(_KCo1 z%ZbcSnV(jv=MAh$F4f6^M0i6J*{EKzNnj+1C;KMZ2{F_AKbT$8Onq|@*Pw~>O89;( zcu##mt)Qxj>eXRXq|w%O7jMa$Ayd{JK^j(-(Zi}9YR|!xe%G-_j~~}g9y)PUyZh*g zLk}Il@6eGWr;naF^NMDe`L^h5@w$Ktr&q9`eEiVK2Od5206sl_{Jye9t_#s;w75l0 z&HS(C7rDs|ipiRnm6dHv-oRG{jfVAXu}}HeNCYXo5~{$^ie=(*qf%9tEoA_wBuh}W z1B2HtNndwrWn;mBW-XR~2nkresD{TSLA!p>af&7@#xhA?TXn?RwpBjkD4H+8=7otr ze+4fPz+HgZTP*R)KDS+7$L1McPbNp7i*{-Hxf%9OoHOs!`8@7 zCeX}PdUR)KhsE=)wm&$ngb`#$QYJxvSM=I7p2CViyGpQ4taU{@N$6Z|aLd+?Rv!?8>x@EotB1BzHe^;5OeO7~ zA!`r|$WdT#Rp;3Ire!H6V95Lm%x#aYg%j6xk6<)I@eD0y${_^_!>BB~44tq%S}?^V z*lBpQKEte!1$iWU%198EXnsLN$Ce&)9aV_mA1poy+}i>IRg}%BV^Bu(Zm3$J@8Po` zb%s?>z~^tg`JUYOPwaiIk~u@1L=EDIUf-yaTa4J9IgXY>!hw$lM2@3;nY1!ED}Ndp z5IM!*5~j?Ujn(U+C3#B`PDTZ|rLX}EEi;gOV`M_YnA9n5xTu^oK&!BDAh6@@Eb{6= z0k(hk;lKc?5-1LL6l*n^-Y3@xn88s_(2<{UidvPU5sJl7+n6tuv1|fstVnoHISSh}j}G%JDITJb#OyJrR^+J<(6CTx^z7&@vYwe-|9zv2&sF+4PF z4IhJq7==#S&!hFS?2MvI#sW>%Y4q|p7}XqFYnW_jluTt8Sjhb~w?xT?qh#y>)G?p) zvoxVEVI4T^*|~_sNJIh&$!n-15mNuLrq&w2Bl=d>djdJM5*B1vl%&E=J8q>=hNldq zW|U8?&O8h|kaH!}#p7fO59%$(rmShU#czr75eurnNeQtBfmu}CYoc_dkrCo(_GF(& zPlz~*)qj{wmY~i>w6;gIxcE>#m33sg;knarMw~V+crw8{926r~VsSiesYJ9-ohF)C+c7o ziy~>H3G{8HPH%Y%gD-WHD1Q<2BPxQ|YuWh0!%V z4%nZew({9vM6rK|LZW`#zl*Tw%=YIHlsSX_dr|QB5wt&#KSJui(BmHjkL+ti5{aFS zI6$6993WRD4!}E!f8nA;zj4nq5jC@cpcy3}Kg6W5v6c{F-zugy z&en5Txy7R4c?j_`Msmq&2S=88nj+qLaHAz}FD2K-c`zO?Ywv;BPkdg&q!^W;;lk-pXQovaZ*RD z&{&oEwn(X?n5Sf0{Av!CaCKtZRjyYlfN}jH~lping%mQ z-g&~|W_0jb0|UPD`I#bzwZKoWFVC@9X${_q2OgHB*=h`ja!IQE>@b$)#bThW2<$J& z`sCJ_TOwLftwvLqptJ0VN3stnCgkw;f>y2cgm{8b7w0(gSme6i(dvv0?Lqk>tLh$C z@#NVB6|AESbKQ6e9z%o|TBCOm!oehdiW zV)Gne?N`2e`%(AEhm6Y*BqUYDArNo|ZiIe6=C|TJ7=CsOyu%cQgOSr+@ajPG`NpjI z)eFdHL@_98H;ft0nYI4T_LnC1FTgDbz^(Cjv`1RBGzrBDalyWikQ%mi%&har+dh%C zCZylh7GdiPx~v)#8_L5ra4RQ6D;B8N`rF#RQd23!wC7K*c^zKEc^TZAHT)l+xufBa zD}|3bGb&qF+OVcG*76?K5iF^!V1ReT%^jX}=u~ww=~072Wg76b#3X?>cEEE1^|gXk zowB2!FBs&4O%C!<4fPUM3v_3HJF6~m{)cl68iNwU@HqvFkvj~g31q^P`Hs2Hru{-I z{!m~gMCD_xpOsPgM$>17z>TKi_hSr{JtL@U`@wZrUG>1BFYi2g{qf48eK>fUnELE* z+cV*lU;hRBq2OuquAkpy9|@oCp7<^M;qd7d_iVI}MNc2T(S9I&I`;Ta*+;{thrWC_ zHnG4XgDi*DBH0zppNF5MxYxlhM6RLYqJx%!ndCa^IB#pd1^%mkzg6Wu3Ggy;LD6pD z4@V5a28rPWQ@BNa*=s(Bq%?St=a@?BA*sJ!(Wh^u63_^v`lFeB)BnQL6C_9`Uan0@ zLz>v9O{k_!FX8)^zm?uq;HqC=u5n5dEz?W6=mWx{V$ps87M%DnB8v?anKX%&ibh!Uw)Y48E5Jq68rHq{t*zduo_Rr%F z3S_?*zciG*55FM$e);(+`T3yyd`NzNmVai}>J9rN2nKvoPrVGESDz+lUS4uPj%spZ z7`5_ysf5D!Cy`q=zSp(!P2KJ?d~ZGO%+oG{;;IaHOd*_pJ~OA-1J+W*G+bYX(;bIJ zJtj|$__oW|>RmOJaQr8!Id0{X6p735nPwMcOG{4kp_JzDUSMfR5xPuirUHjKEtrSM z>O2{#;Pc~a{mV}hJ%amlCOMkmx=fQS>W z=5x)dnKd(Oj6_YJP0dE4reB$r#9n+Tm2r|uTyaQtJw|?)aL{bu!GCWnK9@)*{JlPP z5xWIz*W@C2%voNpA_s=#u+xuE3~9!ITbl%KIW zPOk7^$2vasuQolpi}V8IPo^$%w?IxzdWz>9&Wy+j&)6a!=Q)Pxx+J3SPR&~a(I3l7 zqUVT?o?>GR^k1$J@>*#Tksaa%vCpI~Qn!eWO^Sgi0Eu{tA+$b;(5F-Ll0fJ;vy$jJ zLOUMyFjki}1r(eET`W{!%G`p}s;LC6Z=^0>w`lF0gj*W;Q^vns#9s`v4N1(to|>Zs zW?#!nqUV@RJS5$T3=F6DX@~djg>%1jG`NgE+$dXmD}3F2AV`j2??rF)lk1?Lun+*( zIX73#3CB6d&DqRx4A8zLphKw{O8`2Ul|;`0yJwtS6ew*{meGZZ1<57jZs6eg+;E*a=sG*Rb?a+O*oD40TQ- z5d387B6kbHHq10CMou^rAc(U#0-a|1?@rB70)rpRN+RkqKPoU@Y2(Y)IET;By685S zN+`g6CUqgZ1@4;33b@Q&z5~s4nD2i&HCqYzelshHcKD1G%TkJqo1jBW{Ypp9KqBIz zN9Ljt8K!A;)?Jr899nvNGFl4szfxCgw+&`g4WndGP#nd2`CwAc{!?mJ5{={^vy$jJ zJYzf_(r#!hqRA|^6`)QD1frMp^=dM+b&{1Ca0N4hnw!Q&@unn5yHYcg0BL(x5hha6yXL9Q6sY@%E-G&;bhjBUweNoeg=U4A1-&T-*oZK@Amuvn>U zU6_K)T+upj)y2%+Y;l)pjcSp-tx=L!Eo((r@gYvM=HrE%{)5?1S!p|((qB*Y+I;v6 zf7GwWI#nICK4ebB3xw4#Ug(kpX$He?TmJq@kFkcKjIho{SmzjP>@FhvODh`ULPASL zL&?5T%v#^kqTFkxd91c5cf}Wgwm#y{>s9M@)T#lu#-56C9Q=+c1e^DyE^oKkj8DR8 zR`HbsUpeuhjMaW81^&lV^OKNG@61Xfl1)F7l|-ZgN-Idy=v6@?E-s@1W{{8~AXi@~ z5BGw$G@mZgOJn*%>T2t@F|D7pO4tDyHJ;&Q*Zq9z+!Iabb4=nznSZOnYW-Q$Xd8>< zmDK3Bg&}b=gc8=~DB<@#0bj%?_qPr)!OJ*^fh_*Ftd;aH_|Sd>{dtD|yb*sMcF$d_ z#sI}9`v?qZ!TCI`V{Ghwyl`o{NVm;p>m(^dsG8jVT*yqk^>h+F-$IenEe0go&o{G@ zh&uh>GKnibNZfkN;_VTzz-mXIYVi6p?I1scEhBVti>n=>^TA+$FVfmP*@EcUjd&CC zGtv;S_iapDyWN6)+vJ>-4Yt-CP9Wg2<(#=3)?aX}EXtO{_5jbwRi$iXO`@{3w1*DH zh&`z}PPB)cvXW?T4{17y;vyP~M5<*V5pi*up(v{jc;6X1)rq$4u;y}vZl%Dy_bg8G z;j?d7IkrFwY#OwX3`AOjQlUYf0jaO#P|E|-x@1jWJ7S*BAvbEuo7EmG^iJZVVN4I`C z6`YHR-C=ZGD5YMojK$WD`GP13^uM3Fp1Otpu1Q=C!R2J5n3R(=0#!~5s&eLH#msRr zV`{-zvVHt+YNiwI<1s38Jaj;H`2erc zM~0ME*k2$m0sbFS*I~E7-xJt_TFM%TY7d}OQNt+#kIJ#4#-Z;tL-e!(XuDu*E{V=xFvP2i58w`603^0|L=Pe&ZI74mleC}o`kiX zWXY-2Y$R&>L{<_}|9M!zy3(Es+i@r>9W4RnqRJ))-oV(lPPSSi86=ic7p_~Fc0fB2 z?s86*jRcCjQ2ej${MXJr}yLP6- zhJtsc&OHIkJD9|(66=a9WCQo(v5Qgh5Nr19I!UTu zOU*^1hCh*&MAYymt3hp#PD&*1By$6O(kr6L^LM1#*ZwEHzUJ@U~b-qQ*WaPKQ$;czAq3SF~sH7byBfsBi zGIBh1u8B7G5R+JSXutB45smsi8CV)PhF;?59YYCLhtR$j7Xb8|(dLnZz!FlFYNA_vDb`1Mi=-P;) zcBnj0XOWtfR#ASr!~*5pQWv&cD6bVL7vM(=$9ctUP7j%l;n-oh^ZC?#C2;(KtR#Al zSalF@?I$5R)mTdYPW(ZxcJa^3>Yp$ir!)^(ca{Bmkm67YO9 zD~X=N6CaWeojLQ!sz)7NDxpC3r>P6sEo5!8(RDK&blN{ooqGbMKg>#^=O|4b!d+bC zI-%h#cx)uc#y&c$k`|C)M5h7IdR>RfO81`$f@=q_jJ}&tHvB!4dQM7^YbgP(2ep+T z))_l!zu(P@QRF51Z&^w79Oi?;885*);bAA7!;nM9VaLT)-QqzMEE=_$_Og)>_;>dV z{#&tqSI&$<$#`WD_sOc+87~FlqNAW>;TlnF&)K?}4m-7PNX>4d$6ue7M9-VT$Z;GZ zM$Lfl`+^?SHf$!M1;XR03)HO)9-V}zH*u6Bj@pOK4qL7tO3g|Fo?}@_w8N8Ok{QiY zWhRq>M8w6htRhFvQ~9vM7jK^0@mRx_3F?0fC#cV+#;9|Pqat>kpnj#(1oh3Sb4@g^ zXPCr`I6=K^t@<5PNXPy_>H>FJ&3>D5wZ4-f`hBT$PgM7NvXY2;(T@vaS9+$|$NPHQ z#?j!V)c)sEGv96PZ@|H{e0ql;!UQH_MH z#lvPeqeSa8L5;+>%b`Z#A5S%cQ{7aJgma+vou;~@sf&_mUc*^Q^t^d(dkpvK;RHT7 zqaUWl8*GoD^Z(=8hrdc10s79=MeAM|EC#&gip*P^Px<p)f#JxA-% zBQ;!Eql^#A4;k0AQ1`|Ov$&0V4o;2Bi32h^|G)ec- z24$AXtP~l0z}-5@02CQpUC4pg2+!LTYp@hK*1bE-YadU|dZNL+IxC5uH<;b0jRlL= zW5{}iEwjgiVgGcrS-gk~VOl$U*Ix``$ zPy(>`WhK#bz;>SjKGY*+Q_x9c5qpGkr*S&C=T!gVnRj}c_?@+O${R2B)$Z^c0rc~! z3*RlEH{rZ0=3X5+fa`0Cdpg@3+~;ueF+0`8RrU@?heMfkS% z*tR(7nL9T3YGO3pDr)0Bu*11UGO09nLwrGc{)kEN{z7qUTsn9;M6Kc>R^` zc%hR4>o{N(4)+G!j^y&q)_&L37lNsty4c-fdIQW9if>YS_^d$U8?N|cFd!IwFLTbOVZ z>vkQ^VR>_ERub?$la)lz;n~S&aPqE}OTn~yGJ)&EsSDUGT)X7_P2SnEI&2+I9r$2s zmJ+aiAS;QU!!|(r2yVP@zXdDyUV!;R>LPRt%pmI_!b7YBbvQHN^Qn1AAo01ZBzlg- z9zw#SYenl8ZfC}BWHQgy@e&t7l`IIn&Vnz+cY@?Ur>=r-k-V9aY$*fT+fasFP=-#T zvcqWu-%ibA0_ksMCDC)FM{phBg0U1W$+V4RiD<#?>Y-lE1V$%e(g#P;l1N<9=&;#; zM`~6Q@Qi0A(Q|mVwAFtrI*C{T=ib!C>9$jiw1X4JW^3{whPzX9lECK9tR#Al&9%Tr z?mmZftZLG1SL;ZZODxEhQWv&cZut^0QqEA5=wL8UAz=*+mf(7o0_ErY;Vp=qUW&5odu-rzLFuK# zz0%@GCc{Zqa!qYl5}T`2bCSSjM^+L&$7W}{wNUE_((1_st@~3Ku-iVhGhsdKuo8N8 z6196%bCf{s?yMxfS5TWzUBGTpBX4XRDjjS{Tt9P75}{J++!F}tSxNL9q0wU&?)Ju+ zq*xond3lMucss;PPx4b2sas6OaLh4y!eMq1PcejEl0;}FH7^N-p3O?4=LjW5n(}{b zRsL2_CV2f~>H>C)*RG`G3pk9y_R=J5A4$zp0=5rlCDC)(b{%%Ac4M9n9S?@_jw9A3 z(g|>1N?pWmfxDXQ8Ny>^&rPZt*CvtsVrrfe$bBIziSHfc{xx+GyG4$Mtpo94>$)UH z-$|W&0;7MqYYiP1|_=bpgm>Z~NbcQ87dx`=7|QH!zS`Xom8 zr_Mcr(Y;wo^cI&!bqIHSxhF9CNLCWxI~e^@>LRAW=#Dr>HzqOqQtI3j7=1A-iJoKh(zY#6 za~2Np!3``GgSNl6PMR;1R%G7)OdlU1D+gB8?u@UR0B#QrOU+UOwwGol(R0|Q$mtD7 zW?2hR8yAeg&NtlnOW-$Tt{aH!$9Of~C`9U%m!x5ja%wB2*v{M3su9CV?-9jgKA)3M!-F@d; zOYy$ACkc|BnwbPh^I1vs9Him!IvfWU$hIRq+am>>A5LAEZo%0SY;l1r2XhnaJvS$T zc`h|031IxJBzg`^G-w@>eFmC3=rDVi`^OTI(w4{{Zi`O6A1lc zRuVl#NZnRDaR>*e=K%( zi>vKU^dG^yb4|I%xHYO}!rmGs8T?h&YNeah`R3azaEtoEyj5GS&Nk)#-2JDnQTOMn z(yxApx+F+*{9CuJdU}%1e$nB!83Q#?+Aev4`udrD$qw*GS*tlZ8uLq8N%XJ-NRzud zp?;(8w9{5;zU^FrOC^-%I6m4d75ti{3Qh>a9SvN@#+O_YywTK5C8XDIRuVl2@8zdt zj12`A-Fl2cw`J3boOH>Eu6hXZ3n3)n-_Jj%)80OuF;=l;mvo!aKh>84|C6b! zpWOUw{X4$J|qF32~6YN^<_!?o=MGB0>2AcN%S1QoriIwMnn1|SyO_3+txQ% zPbSEHFm(aDMQ#@s;6e_`S`u*>>vp#!Vf#R8mJ+bNFDr?j!?x+n14n}4x3!zK#0WT_ zPhE^|!Pz|N&6k1!GTx2$C-L}PYAzCZd^RhIp5rk}2CJZD^5tW#sC0-IY`&ekNZn#H z#+Jr_C%#-Pj?jT5Lf=ZwO9G*9W+l;cgtjVcRWQj;-ml&Y5 z)H1m^n8ay3H8%;IMzfOWIZh*wnMPGBIocw2x7VH8y{#Q787&Ciow`8X_M?%>MR1fW zIk`mxy~wG>d(o{)bnZ;eNdleQvXbaII@g>jTk~eUysVY1xjDn7%`AEiGbJ34*$UJp zQwnB!>SA__8Pq++XD;M37xEd?8*WSD_hf3W68Jryl|;|+yDpfz)090@iNm3*6~oJ> z795{VUEFSQ)WS(PO}-<3LnScY7jI8u`R3G|C9r%ZD~Wb2)0i9;7ty5^a=SqW5)l^{ zvK~RMF0D8Mi}9-0D_rrbeRsfd5XW29aj=^9aHFLmQmvKzqLcy zfV1pb9(#eFkh7Jwy%W=iQ0FL>lr7pcqU?T<~CRoB8H*NUeED0@?9p8#blD~X5Rc0 z18u$K^2r6Sh14bP7B9KiS2)gr;~Y57DUSJy0I!*V&`Zrt0>T=Tco9dB9L|-f^PN4| zudBz$=lBH)WFsymZtzI?tw8pU)RojNWO+J&A}9;CX^5`D`!(WaA!{<1C4pZvHERj{ zekd!6s0;p}G=VF9LDop1Zv}%wi@xP2qXn*CO}y6ZuB!8t%4*6$U)gRuTpSX(u zh5mJ`CKsq)mb#$bc8#6zR3Juk3bBaM1Yo(;Oe6riS!#Q=%S)&1m%&WpZ^UAI-8GEn z+nO;SM03JcZiny144i82efWhJ;``<2r{w2@^7A47@%wNK6V$_|&C|?{q$v^z=bBR| z{eG`#8l?u#Ie)f^r@8?{0`(qx8mQxZ^jTPp=+W>uTh(H@Q8MPNvV3iSA4+0BPJhnf zkKy-M^raHc7&Ds>e;5D!jYS<7Lpjx^KXPam(6) z&GA9XW!659pPm8)*ASd0=iVX!si*xdPX}6_4z7aMGS$V?^ydQoc?14j3bLDzxlSeK zcc$TJ|1*tR&A}yRuR`^}`mJitO9|$Qzgd?8&9{Khb~NoWmEeuQb>_WsG&1M}ky;`B zG6RWrL7IU?#Dxr<8AwE2h;oyGM2rhIyk#H}aq$;f&40uNI@U4jzpDw!RmUoOKfdnR zA8%hRf*hUwsX!cPfj|`ROS%K?GAP z$g@5d1CVPqx!9MptbQ_eUWq~ISF@6c0RDJZ5(?l4G69@f85w7UF$A!|`~mW{ZMTyT zp@|5Z-%MS$&bkJb@p5SXZR)%dX#O>mShcrYc{};`hoIED4ITxoDpnTJJ!qhB+4-C;afpr zIyL)UHgvyVT>sI>64kyhb*71G-;Cq~-WxH5qStd*IBVO5Q6|0#jNN;~@>iX&u8~@-WYf5yFR!}bZ;n5O>m(%NfA4wxD z7g>!RDpx`>{D;&GCIEe7RuWO4{Qj&YBFP}l>IF%L!VBS|G28IBz=Tl16uJ;hl*wN3 zAiCAeF8eV3Y_wul-qN&92!xq6Gi#o0+CLz{wFrJxg6k0cLkX@&@ZTi30m17x@MnDp z9+Tik1kX#bAHnA(IDp_sRd8m_TR|>(Sc`{#ni;N{8API-1UDggT7sJq{6Ps0A@~a_ z$W*>655suBb0gCj=J)pm4@A8x4_lDvZ4w+o@Z%C3Met80IELVNRgkDO!iw5Y)rN{% zi;61sBcHWgQ7aOpiu#BIsiMB1g3QR80ZvV|c9R5AYZ2iCitsuRegnm?W5Vy3AQApu z2@>IdkRZ|83eP5_B?Rt}AR%y0fz7>cOW=`lNUj3CxUqi?n3a4 z1g}PLNrKlP_!AO*34))J;7bww8wp;E;AWii0%F%8c#8xz1Yae=>k)iff)fb7SAvrW z{|LHe!sg3D-6^=nY{M#-s6F$i z&*7IWq|dWSGkPN1X7of@4DW1FnCAQTU|4sbmz(x*Q^i$BnRmR`m)#VFTMqe~aavHsWbk{DA^Yr{xl? z%CI8)Lj}F+R1ZgQ8ngaJXV!+1knl0TI&)RSAHWze(lqT13((ew$v z2sb%iKqkQxaGN8t4S!v+R-@GRo!c>QKZR1;FGr}g9_&}()k*ta`s)vHnICZj@E11sWBdbC6Z7MkkA!pRd z#)?Cx`>>Mje*u%R_#yaO6vB($g zC;TC=2J`$+uUjs_1BMgOqKP@~ zZv;&oj#+N{8y;fP_2wI!_FIYkxP1{n_6q*I73T8Uw>GhO;lLt99Dv_t+y<+ju;lU< zB%6wV=x;v3HLtBg_lIJ1`#d;~!L{Fh2VQn0w&@I~&`#;t6rbSnM`N^F7_;Ba1pTdQ zd98&d(l!RETSj%v6{~6&qV_p_Z9mT>Nbu5=aEK?OtjI$W3z}H3X&8jDzoAHrL6dt(u6Yr#d??Nl?vnnVlRt+{Oh#PeaAdMWu!V*ay zOn{y`VioJM^z~d5(xgs_1pIfW>HM5v`LO`bB><{?@DMMFY0qmZT*V z0%G%`KisH+YokD9Y<&rDM)eXh!CjS36)L?>mlLn}L*NQ4Rnd#hl0}$c7_yC7iQZrh zuM9V(=$kEOy+ZXiAH5nY!g--um@}%y81YddQs9e|>y z-_&r+1+u1-|EBq+y<;9cajWNGijTMRwhtFz#bmnusL7{x*jRmdamPDC*xhKlhT zb92@b_`^VL{<8Ti&-+`Egc=CY_X>-2P6g0d!B|a1jUsapG$)bh+$Ab?3q$4uhrYb? zr1rmG`RD#dw^5Z>_gz(e!>99~e2w3y&TThe>koL1V$r~$S_0_-!*w0E09TQw|C;Qp zlm*sirm04!U4xeCylD)wymkm7N@h43cxZU|D|8}ebz+asR#E1pPDwDR<<;|fpr{?G4j{`Mbier=qmiw%}3EniITdDDkf zDc0RL@~)%*{nz4O^t2QQr4$W+SE7ADMIwc#)VyeaZ7s$Re+s*&%2tubujz$qDR0ZB zjJ&k6+=68+?zOSN-SoG|liK=%?iF3DR%dCpRA-E#5wV<`_7>*%E>$2u2jsa1i>p5< zVY_DFVTh#NGzq5?IR*Aqagag2~ z_*)AXnpUwcyiAhyns^LJ^%8XUSo~N`uVU~o6g~V8YRZ2Jxf>MfBr~Arc`zUqjVOMl HF`NH?B^$sc literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/.doctrees/glue_cache.json b/doc/LectureNotes/_build/.doctrees/glue_cache.json new file mode 100644 index 000000000..9e26dfeeb --- /dev/null +++ b/doc/LectureNotes/_build/.doctrees/glue_cache.json @@ -0,0 +1 @@ +{} \ No newline at end of file diff --git a/doc/LectureNotes/_build/.doctrees/intro.doctree b/doc/LectureNotes/_build/.doctrees/intro.doctree new file mode 100644 index 0000000000000000000000000000000000000000..c280af721470ae8b42d436532a30f322b4634e0b GIT binary patch literal 47257 zcmeHw3ydVkdEOn5dynBg`8Z166V-A^agW@YJ@QD34^h%Rc)a7?EzfsH@ucK3=$Y=> z?%tm6VLxtnBrssiP$un`9Kd72f-FZbI5r|UL4d>#93ujf07;O*iD1A`>fu++ed| zT2>HRc09=TT46iw+Ceh-rNQyPJa}`klVI7T7)Gv*_$k zTh2Y4E()=SWkI(80@bxrP6CFoqa%e16HlPO6`;U52DqJvfLsIrej5KC#=nmPCTGbx zKk>$^$cNE|zi`iziqVkT~! zxNXxnx_07(R&4Zq6EBT6=n$be3Vqvf10!y`cF?vP#%j1>T4p~sHf#gbs>Nv(g=t`c zQ(Aq)N!L)^ML`l9VaKq*JPqSq7#a4a+3orEGJQW{O%1aZril@PlR#(R2XCU&meDiU z?Dz$v8=_j`m;u$OIbd_@@Rn=Yp#errdwyurSCi2AT$o-@Tee}}vH=0wzycuAs}rS*BwA- z**;-XV9)@8{MhIKjb@sJT{A%!l3=oqDMWO+#>a_Xfb(|Z_c3rYF}gx(eId1m>2~qq z#OOsKunIgim>@4XN3+l0K86|OY<^q60-^Y{v(wq-Jhti_Uxj!a&34w*x1ssQD|GG* zFRM{xcj}7;z)r#j^vUsyI~c+~dwYaVga3AYVQcU@cS;**7`Sua+!{{S@1Ww+&~isV zyEV)_w{$9~@7 zH2dd>w4w~m`E8nNxz%_yD|nOR$9We}eyNNlbU?uNHZ<3#q}6rfXLjw_!RFI8%m>WK za>`qCZ;@Tl4^!ASvAgELS|Q7JJqR~^+aeo=*(R7YgFel|WFknD$n@93$W0u~o@tDU z7}n8j8@XX%Vn-i*9GIIi>w2&ynWbQ{k0%ZcsWrziq-7Am+>e-XCMv_22*RifV}Q)K z$c8mVAZ)p6!4%;3USp*Jf)3S@% zTyc_Gec~6(?Q}41SdVSkcj%UxvZKo(f&(9Ci>D-9hG_{f+Ha&Lgbc`GxP}DuusTDZ z9kWe#xr^=yZTZk(_*yW%`B<3s&5w3y=l6M-5q&+BkPLV79qTOK{_i8Tp|RlUD$GaGCcP_7b=2egSR?nhBuJ zMpvybNVDMsG$KW^-vY_LSB+#x1j&ArlWaPT|2k+a8+=#L_&21r!)dA!?UmEe@Mh1% zq)#>soO5wG@bV@7f@2WvJDCPYl+(E4#d20!o#iOqC5*5}!!LfbzE+WuiRwB0Mv z_WiPGlMQC0?e+6;*ns(9!dP$&CSXHC8ZbD;0vwWt>70Ap`d0}3cR=XBt%lJ11VaC? zEJ9_2*$92-wH%RTt-%=0omb@2itII5lChL{=G|p>2q}SOSc75L2qVzIguw^<2v%VX zYYJ_a;SMy7OYjDV4CPl{GTp((m%=rsi?qIT7XE;YMiJ|Of>{5(8nNyd#QJY#i6tA% zCf0B3nm({oasmf-vJpn>Tp?Jt-7~nIN@5{Rd!nRf@ssWzS^_Y`77ZH4OVZwlitpIa zKxn#QMixFZ^tELep24lqhn*Z^nqd@<>^A((WPy)V&dK{@JN}#@yxPtkJDf#tFa6S7 znA4{Qn*JZ)Y~d}4ma`VlG@Xbx&TP(jRp$)8KDMx8V1k=qF$!^vKz2wg2x`X(o?GVH z8ZilRLn{oi)u1>gr`X*4Rhax3$K*#VWAdoL{}6>Z$K1C2QJ8v`W9pg8n0ioP>SP5>i8duLwQ|)KUa5R9 zB({mO*sO5n%qi0LZ9j#*5`njJ%v5ftF!3tK#O2DE_>jQFD-|#y+LXY=(kblxVDkuG z32bDMKwY)n08G)g$2d&G$T zaDRaTB63ubQv?67!O49~(pyrvtw~5Erv^+3@|{8c$U_n`ix}IFKy*MpGwi5#yQH{$ zGgwP$a}rj}x@}zP!y$#;$;deK_~S&875sS)q6}#@mc$=Vo;Y!ms`waPbe!+7VK3^> z5JWlm{8<7D)9)863&BqbA=s=S1fq=;0_BsJK6ri0|J&N~?`>}SsRr2*X=2rL7rXN0 z=P#B2-PF4#cfRzYmvHAFxDvq;iWOuG&q40WA1Tc_!}|_lWYPN;e$BE;D6HX_K;*Bn z`f5dc<0@hqPBZU=^3=nbtqUm$HRegjrpdC6nS%LSA(6S8ISguI6#`Gb-HH&Pa54A$ zW0jZ7^$PXHWShC`r5>m$Jpkt@6i*ne!66URH0(kZQ%nLiB#jcWpb0*NSJEDE6~Tj_ z*sz7xYiO*sX+WQg6*>2}LdmlL2lQV9?I;1~+L?7))< zG(lJeBl-B}jinG?>VOZ^kD>LCbDw-o&=4plp!lkR0Mh%AWK>}fXXe0L+9u{)x`WvK zO`%Vw#4;4sd2w4@uEDX*jw(Z-g&>@hL}I%`Tb4Bqn0fWMEvV)r`Z`o|>Tz38%|{dI zP|c~w?LsvN#W(`hJbK&?RA=U(r!|n~(%p98IlcWkLjz|Xy=@1YdbQ~*x3jW}4GYGo z?8zbnl$=ZtX~x-@4SJwAWEU#?bL_=PD2XunoD#OD9!NpKnD9;Jo_R#TCb>B}yJC(x z?><Ld2{kj>Vn6qz|`uWj;1P5qh{h$}bGn29#D9|mnm+YE9C ztYRw#>v0Y%1qN61FFr1yeps}eg(>nc$~GU%+c=XEaahJu$fb;gb^L__=QA9fPgMqI zS%C9GKD6mBRsu)1QQ%CK81JLhx55JP3l|z~Rr3h!yX!0-CkcC4v+y7l1oXshguqjv z46N8NLhQ?M+y>*2F2c33q%hzRZ(|n)LB&{$plxEJK6XlM3JOnb9}bbLv@D9SA`$^K z+;~n*_%if(nI$VFUmCNVeID-49ZoQwDP(EI-n1k5eK9`1Z4(Qt8=JX(!$yP=Td6P) zUB2xL|NY1o85TEPcN4I&1)2;#89Q8nxE~pACA}v%eu6NBTZw3$=?GI&@_FLCMel}S zK)IHf&he!08=9&UtUKkLLq(+|4nkCHrJ!2pP`UUuOBs=4;e>#t!|E#$%pQjPVcWv@ z!wtsz!*VFX)ub{iS6L0R&j-0dzbftLJgE&=p#D3EWSd$I^6WQSI`4zO|BBT_pMxAhwegJ0$t{DsOG zJSi~vu4p-nNl3-ZHt*zZrXV@E=F+hi?AyR!?9-N)i-ROCsXOdIv*Sx_7n=> z>R{we%sSw@9O0Dk^neO*W%EeSp~?#Hclxu66yN2f_;zJdJS9l+*K)K@*F8iE*`^dJ zo<4(lhqkNpDF_BrK>!j*HIWOVo^O|Iv9&_X?{dWay~>DrS|H}{RzQqwQwlMUz8skr zJdrp9Bo29D+h{!AYFX;H()AVaKj6Us%gVq%BY^)G6@Zs*N&$aek2VwV!yjr!>)|b< z?I0)@Li@7Y!XYdfi^dq_vJJ;>FGO@){XBf6U=>v%XPVp7d5+fX_Vf=k;EyU3>Jx%c z|EU6@WSdfi`r=E#GT-55(Q0g>VWfaVT0{vNiQRSr7pLezj1(d5wEQElbri;N9dnus zBa`XdmLIm)K{P1jFidE7u-BuRo&4Ay4u`G>ZU+V^mxr95@9P*8+27Y;fIEIvnLC~p z-0^=aaEEMDiaSnU!J4G)_MqO+Aa))*7EouW!dK&?-%G6nG}5RI@UCpAzu>1<`UAoyn$@Z9bQ`nPT4GSRIZ@WIBdLKuQnQ zGEP$=07-0@mX24H>~i})pwPR?(R-^hdOt1D8|SE+ZL-TYrO^BISrI^pX`Dw8UWM*K z1R+hS{lm|x_G`Eq5%d`2MV=PMvawkd^}2d@=J1xnB-zE{Bg zItT8%m4Q1YfcsVj;AERpz&)BT%H$3;j52>bWLpATy1oMb_c-vsTN(J%0{Gvl0K9Bd z3ix|Zhko(}oGT5a=Puu`D0-zp`U4Kq4=RInMu7DD6+n`0N`dssYx-GFsThs3E*!xW z9SpM^c}L*J=GjjQgExi$M3Q`-T`Cm*F-PHlsf@z20)_v%0t#gtg+gU_745B&ttRKD z=ko7pc8GNzx-qVk1Ig|p445JwfnDa<;|}>DH$5J3m=`fVG+o)0r-y-`o2ag|9AkREjAw`LFX!R)~ zPHTF^DV?quj&Bv|az-dBQV!(tt~FZx;-5_D1K%N~lu=tIAs z#pBjgC>w&z;34W7F&#Chm??HreGPs0SYn(f>nD-iCPI827G#J@gALR%aP~ZQanwwO zQ$4*U#Ay5QTkHT-zITMm1?gXk>YSYiQ8WPN(_dwHv;eBX2a-dS+ zAp8ugug#R-Kf?jkb<|jtKc(d&Y`DlPwhUb%7x2`!)&k=ny(zU(GV@-ebl zTTHx^JPaApj2K(zII>3j=Cn6YwQ(+hyd6-Ml%fnrG}svd>fwCaYA&lh+ zh+)bj3NY;Q%s?*AHdt(9x@8A;2MKq=P7UdPQmpR2kS{LCOUx7P(8o-~57<;B)tj~Pihp`aP`t$L{bA|P8}9taXN=;FrA%; zg3v&kPk-Bi=<58nl?LDK=zTlT>UvbkhhAFH`|Qb7N=eaVn*I6(uJ)OIE+!-dX&~7a zO(4>Pg2WMi4KqUm&tk`fIqkxpW|NhuOH-1{*Ve<9Kb_oV!LRwLJDBcfQlmLWG-9^A z3Ih~hmyKDrk!H{B(VBp14$75#y!3cALRV-64F9hYRa7K@1$ADU3x2dw&#A-85NR5v zQH@&K;H$PQ80GNnu?w6a=MVO`k`VrC(L20Q!PzJmHJ=l6h@>c+Lv|6%X&+!5IL;Zf zWl#35$=*mz>6GozYXCWy6&!fzA=@^5fPD(NKtApSwyJ?gb}k+-*4>2rJB;oc^Lrq( z6t#iV&>~Y92lvAbiNf2`x5C(M=^i|Cq6&!}T1wm$^sS4Qvs@jtX)W74R{eqtVwm*Y(@4)vtgXZFIfAa zrpQu`qXA~u$MFE!oO#FD!p5?Y?UFp@wcLwBS)3>&Q8>MG)%wZp3{sPp?ezm9UmPKF zkpeELvsdCt8;dF{=yzDb+)db(;=vyjKZ}=(-p!G*m9kTtf(PZGnU49y773|Ax8*Jh zkTD&}+MMvAiHxMsHMvf~muG8Hi9b!Y%4^a4iLO#G46ZIuN;((6%mMM#r^H#|=q;_N*4K+&be48X1VMXaCS3JVOKv8=X_in?QgY!cVofnQ& zO+7FpZ39kLEcAO}FC|avYc1S3kq$LnW`-TPA($&ym~AW-FtZ|A+g*>&Ok!6X{&0Cg|HC@2gc0?(M+R6*!P$b_<;a?L)|E-j~-5mhq1l7e!2JplKEDimPK z#u4cM{gR4gkKNW%p=i2Oj{hlQkJGP`&WCqUU6xTdXvaeP{6Qz%Nr z3DCKv6xS|MY>Bw@M)XI~05}(B-@?#qL!Fa)=Ey~gdL*9r`?(Q!PC@;{x4yPzxHT1X zR`s}bJd%s={|Qc$0kuV{ub|FLb0L8S(40D~43TC-o!T}-XzWt9D=00L5X~Q`XoIp9)M$-fuqdBuOuReueHGiyNjWZD22Dg&)--@6*1#wuXPL z5aVy`9(F>|O<`az+a@X-yWb;l+pB@@8% zf0?%5EEgw$v6%@T^!q%k8u?+hTPbLRkKJcM_q)NT$}U==iHHujx*c-)Fb6CgMFIS< zEMp~g0ZH}4K0P*!E24@BwuoyKV8QFN3)2Xof^K-Sa-+#)rm6P(9-M$EjTa_X0f(3) zOjf_` z9WR1@Vxi-u^l=^E(BRn#Eif*@vvs`eS<}7W zkpw52(~XUd1~eSIy(nSxH}oq7!>{m)E4i(FWULPj zs)Cn*^0!7%=sk-9Iq2+owJ|`+RcN6oTA>k!{M#c#o=kP`JgQb1l4c6@1)E3&=@*_& zit>?d{)QM~nMq!@`HhloXp)z0G)7ZCvR^?iKT}NI!?QZAvb0H;CQvygYpFViW+2q5 zu{w0YmL1%3k*$=bTR8di2qWqicW&`$Mt121zK`#((F?JsfPSWLrnXzzou#~*9Q45$;1wIuLIt0;?FeSJ@2R>*KMA!Di zly0=cB_$LbmnI*})%rsPkiZurVbKVDte|ENGfir{_6#rPhSmLZQr%ZZi=3>t(SzcO zZFB+<_m6-$YkJtm*R<*4Ey0Qs!OW_7dQ04*(Q}I(6cD;ddCIYB{;Q|9#HfmNTWeo1 z-r2%0F|H;-)h0W_ec5%*RjUp$h?mmujScMa6BE)4Th255;A=4)4Zd73=P|cZEGE9s zG3UKVY`f^afM2uB2J#ZWE%@zOR$ucHZ=OfSKiZI|i`n4G#c~*FEZ~|g8XwggKS5wD zxjcr7cD^wCFms%i((*1-DyGv5eq@^#yAnRKk4sqJzX2#@UgXHUK))(w)%YtKT8mw8 z(^97Wv*v2rz0z;y(+ipIEY777Mr$iv8#uTxR@-{yyn-k+SW(Q8vTSrIk5}DN=U;*C z$n5jOdbb8n6Jy8*RZGO(2t2 zc#>Y9&zW>G6M)PPxp8Hi*Yh@0-1mQ}d7b$wDZ)+Z5gzf@==IqF38qV=-Vi6oVlM4s zH;|`c#N`bAdJ@)2jWZ7%?aZ+|in%IbW%=qN2}56mr4EaIO^ry3kzSR0zv3nIX3>)sr13JW26BJ@IOZe_)2ifOQyT|G9W8o)t2B(< z7cuRlnp!Bt!t3HJ54_?0l;FH*uAsebpC};MY+;jF_-f7zvxQBzDJ5)QZQfwB9Tzlg zm2I*|UC!)a1+n8eNF`=(u_Yom5@@#qlj}O&=8cn56!6HR$g!$KoCvnVC(h+OZzr;< zc5r7lWmm+6j_2%qavLw0jB?TX{OW!?_D8ye~=@amJ9`gCo2>S z-k+oScLpXU5T)uW2uoE5_SFo}&TuKmNezHwZq zSRHo+p@!WK#eN{nXWAXD^W9N(Vw=o#WlX~gXAV_LfK1YB2e_}&W_Jk+cjvGaFR;N% z{ufV*obvUeci-L$D;GI;z9&TZkZ3$+LCd*QRwx$XA+uGXQjU7kYX!!n8sMOhmxhbW z=A^DPaV&)Ti^ls%c!i1gbr?YRD4}NytiuDNi9T5901W) z%0ihwc^1iV!h~J^Dy?3=InVdOF!%P-7QD;+kO1y>W(lJ@?i%9V*lAx=j}CTlS8yBq zYb-~NiO69RT!du7Pr}0#v~3Q&WVvA}8&3k!@5Y54snEF-bPCrEM7fNG=AMCb3?K=7}=Czrc;U9rZ4&FA(^ng?^$kYLT~R?d zyDp&q)zVONT{OE>O(_*(=Re*NW@LMS?-M&J68wid0_*!Ic#GaBW6xHOk}G~Tmzmkh zQMM_g9J9|gul2{a8rbnzW@hOOf%;F&D}>7?*DDkKP{>M8(BRfnee z!_weuu7_De)|5>lrWzILtH7$-5)h7(V74g8G5=mp-Pxib+msT8pKW51nx7R{hJs$o zI6*q{C~+Nkzwt%8^uXch3tZrACr+f>%Twwta|b%^aVHlaP31$*bH-L9v-5ua`;za5 z3l$}X%Yo+~mF6T(kiIMg>Hp-Yo*+oFLa`v_^PdiZjzI~{?1~D)*>wT+z5B|#Jw|5g zi6hm7rw}{WVDZhJ8x+*HrF$T1lz_9vP(uH|bKaOOhO$j5G2CokN0?t2ox?GHC4{7l zamO~wJ1A#NG*Ml3=rVC^m%}E8{jP;1wh?E5n=zT((E&2%t84W|P*;)g)znxdv+6~P zKzb?g`g~~y)Wq&9LhPOqjVJouWQAg}Ta*TJ0nQ{4Q|lz3!t^l2X-@wi&nPv-@g!^R90kefmwkaiCUv6H6e@fdCUZ=-~8)`u%BGe-oxpLyv zl?yl~AGx>+G@xKZZT;Iedo0->mbZvWWK$c3frLF;@G1#fHqNYGzqm~K6tNYI;PfJi zTVMl}d@7z4q2y~sQIzIAO~igth}auBf+x&uvO=+l9g!lIqfZXF;Efu5!Ar2oyQKl) zLR8R;COU->;%qitka{4`ouIA){Qdbpa63xI*&-nCBK0h!U?_gwZ8?rU}-F0Zg+AZaC+Eb#Wx;;*SVh>10T zz3+{1=CIpnj~Jk2{rp+cJugCMe`ln#sc4m3QGZp)s~pM1DN$A^me>2Gyy`fW<1T)w zfGB@3MNyP|?_c3r^zr={T@StLvx=`2#-%|ufTlRp;_bGDk2`2S4- z#H?6f3Lukhl$x6U9XcW;gU%>ESOl^nxi+1%r3>w}W830j+Ky~b61M3UrSW&i-W)7B z58WJ$2E{px*#b7sqHvJyK>!OS86MmmEUjjHS}3{1AUk}j zHGHG_AiI0b4_kxxvfYkrA#=sx-Ryw)fGBt`+lv)Oi*7|9T-=e>5;wtzLuCC$-CAUB zC_vbw{FwI?219&u48NIjJAUz(!2utp;?}TE%nk`u);H+hNIOD!`A-9V598m*Zw{!J z2dS4uPf|yAn6|tR;f*|h$7$nzWUR=UOcL)=)b^I}gOdVxl=iT<`2;70czSZ^>EzJU zQ$!c<1n)a@Xh`fO6vj|6q-FO)s zfXIk=;605N?|IMQ=eyn~*x%3M=jA`z7wKRZyJc)F9lK`0OgLRM0;082HV1ZHwKvSeLP#_`|ACxj{m<8g^>2 z<@Tr}Y#ijBVlcb7ni&=ao$s^VJeX$SJPx8dJF|oB6!(M&NmG=sX8UCc-RwoG1DkK` z?%Y9k1ZgagBEXLu9YpkCEn^L`-JfOGkSmRK2>N9W`Nfd~CyYqhz)OdV;Bfsb9`50v zX`=6SJR*M}B|d1Tk&mYX5ZPu6DFoanq&@p|+Djb_n`*xR(Xi%OLy6$eWQa zyl+4}vxB@}H3r%G*zk*y#@S(lAqJuxpvVVIgY1BB25TvDK>*_xp6`OP&+elvG?CTR z6Sx2myK#U8(IA8o*J@Hecz1K)?9L9-ZgEo;Wp}w?ssk2wg_}9!0kN!rMTj7Q1(EI$ zD*<+cm<5I8MZMTYVy)(8mssjBZUYKqyQ#NiT+&XONNnwL7CqF&c~x44Qw9n?$~{sb zoX4~M0EVuQX0a{k^nj?=loMuluxIz0c|sN%-hL{ffHub4S#DI3Ux{luw*0W{TPvs; zuMDyWs0H;%Et(1L;6S_!u(}uJ^+!%T1+*NDI_;6qP>K4?F=?waFM0#X_&V$}v;&1t z>_ZeSM8Lv4esOR15inG@nafr_{AP^ES{oDiQ8pSJ-?DK}R2YD72}c$f`5sKt`6C4H zaNG|PbF<0L)gZhNY5UZP8>8iY?3q6{alte4m5|AieFU>21^?I) z7b{SV^iLD5*V7&ba}c_R845z&-vg6`Z8zwE)KdQMC5?b^f#G_GZP*$XM6!=>Ikugk zw+C?4(_C-bxNN|0nwEuF0@4WYX1fr1G+0er*qTrc?j`%>cV8ll# zE&*R+S+>9-op|sjbk$HP<Blp>@Z%)?c$OZYrXL}-_vy#eusFRZ>Bky9TKJJ>`{-Lq zzsE+zzqabUwaPxU%09BnKCsFLdBveH%ximaBM>Y9C2-6;gR2m1Tyq3f0cN2~i1-I{)izXoPhQM86Di!FZG~c@ S$v->X={9v9Bzz%M=l=tTZ}5!( literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/.doctrees/schedule.doctree b/doc/LectureNotes/_build/.doctrees/schedule.doctree new file mode 100644 index 0000000000000000000000000000000000000000..439b87f3084890386a2e6a6fc0d9ca23fef370fc GIT binary patch literal 6674 zcmd^D-HRPb70)D@k2{%4 zI_Las_Sf^@I1~Q-sx73-_xn6aa-9fU`GrKsWhQc0eNo-|dG(@N^F71&tS(I~Dt`tu zVyV)^h`f4M`4yV(RBHQ;{Ja&hQ#$X&<*b~O^Uta^+4Hkbxl~Mj@@rdyLtj2-HkKkO zQ^EF>lPp#Fj%7}>j5}czPb<;^73a6aKJD!klAl)L9N5SeKUZ+<+*R^YIpltkJ7?6e zbT~Pk^qfj8ooYT#xwRO?^K8zu`e`94i^;SQ@dZv8zFgN6}^R1k(nO^=CfRhC6Kz7+|rmukF_sx#s`l>rU)>2axpTj65gb-jugCVN!Vmc9d^{ zDh$7G;`c55z5^S{HMuTtJvuMnmbVE>5({XqiX4Q;=|mEx%1($s&D+J3w_Wpydr+v01r=V|bHY(Hfph5l> zQ;dkUNnlkn7x@Yv;Vh0A5KKqo4$Adtq++-a`!FXS;2SiL*atS%*xE469l_2tx2Dc} ztzG{&4l^Okk74zPrz6W;LzV|MS-zZI_rb1K;0F!6e%MmuC93>wYfm#t&Pw)3U-mCX zX4%tbhuksNSy4JpffcqajNXR7*3K&Sg^3lsG6+yKWn-huqR)O*3L=HofdB~z%<~vg zW}R_|5MUj z%e8A%cEtLBJo(g-nO)Sgrz5bj8jh_?X-fguN3 z2Jf&yFYuMZ#Yhns!?ez{Z%W++u3f<8P6yXQ1K0DD!qo~K57*j#L~*(+5pQv83=!ag z2E}+pY@`fooKx;(J3vk@FgfxsapYf|?#R97$e*3`$gRNfNB$MvQB78X+lAr2tq!{0Wy2|@zaeb_qX1?d+#1w zd)XYR*cIumV5@HKJ7F%eSOqJVc_IvjfD-A`t)m&lSFmzt1pgH*4zvH^bY{QMnEm%B zHG3;?yxCuTq!OV+cH!C@t1YHNfakkh;R;8^wX1mWHcBLe>q+Md`@O+{(vCtNSQXno z?tEDe`%3@Yq6N7mW>e|Wmo7_dFI|(@{ClsljcQ`e*HQz!{CV0gR6?P2gBu`VP380c z{%aq+O`hL?rpe*ofHFYv*J%h|oyz(-KvNo3y$Lj^>iP@v;-!Y?!3^b499yQf0)ING zW3&RtJI3Rm*4MSjz<*m;!M4mYj^Y;g7)SeJS2=}*x-H0PVqIWS88@9Vcel}-RoakgliOjJrAU<8pFJdDz%~8)Cchs&IBX1K-Gkzs54K;$Irbn~sm)a3(cvPEy!3yV4 zEuklNscUkLjLxY%s{GtXJT1`|YP)1|b<_98BBw@zHjwWnxDruvWd&X?)XSj1kniAi zv3V^`innE_2l%lMikn(C|vl}qqBZd^0a%M^_RrH0#>L%OWRJG z^|Of>mSZfpvRC)7FvAF*VWj5_j|Cm7`pmDW9Mx-tXsZ(-TKQh-MjPMVK)PJ{8;3te z!WeBLiTp4X1~}7JV^{fAUAn_sW`mZ>-#qf=h>Gp|GiT2Dm0Is)53Tf#!yk?a`-^l2 z3!qRZ6)DbC`Q?=7jyzQ587z-B&Z3ie_Zl*;egu~DqT`5En zbvG&PWr;?H97_|oJ!>!2y}iADae&T1-q!|wPO&?{OgpIjRgxf$BoU#(29zC8UBLLp zx9_|QSXRs^KC%prOqRLPF5|RDD_RLR3d>L`f+c#658(odlhvaIe-nhtqRzLv?MJpC z^M3BHh21E0YS9|eIe1GLB_Q$zbi%t=INn8jkUPE~Ni~)!exT|3R1?FOEbbE*Juj(? z3&|_ty?%{S9rFgko3TVM7xkp6-3k_?_!B{Tr!24;YOsf{dE@S%gCuC^{BIMsY73Rho*roTZmH|+16$agC=Q2jz33E+BQk;)SV zUc1O~GZrq`{1O%dD;&E1z$T#%_zT3UL!)(<9K+By>Ru?Zpawg}=V`Eh-o6g^-2OWL zS>K~!cTiwUFRJii)5^E-NT21B^~=*{3!0W=^rFI6puHX;*J|h0Tb>>vB85>HFB|H# zV?r4G+L6I-cJGkJUb|`#x_n4KqNH&YmaiVw)qFYQ7vcD!sX`I%LabA2DF9B~HS-Us z*@E#e0Kuo#Pr!a89PlUfjxAzNyIY3VN13`iy_Z$P{93TOLeAP2Ov%564T*Lrff>5rN^plc%Du|?J)zl!bRJ>0(gnSdJn4Lvczx zA+EH<<<>BjUx&)$rONk-3~Ixk$H-RmIt~Ne%`%odf2_e@?%{s7;Qsg>dZ~xt2c@`o zte2E#f5ndVihHlif)JMO8cVNTJ9Lm6xtIpd|u!p-OP=^G5D8qQwxyPj9Y2@tV z3R!j1IPN;zfBfZ*&5a`)Fi|XTcfby7>sSb^^0(3UV5;1Mzf^m${RJseOZP>s28LAsMV$Pg3{ zc|u{k1}1*67x)2O6ShQ zckbV9Ub~G{r^tGcPD*D*LXb8Gx#5mv?2){@b|eyJibyETWaFV9So>W+9NzQa(PE=f z#>T&^*!bN0%uHLK6!kSbS$cyNvl`bDRv@vwdmaw;f6Djnt61Y}-#fk`jzQ^*CuEv#5?%tyVV|n)u{u}lXqPO2nahmPEVDLr# zkoZr@%8?!m`IdR6b6ilkgV;VxeVpew`&luK8v}XDi*#pfuR8m9+jEQpX=1L8s}9!? z^FfLGWSij3>EKFPbKWwjw5Sqv=_F~qizleLA6adnTvAz)|DYwdK;8^5Z8KTK7UfFTQ-MK#T7}0-5P4xn zbu)+W;aNe+9_o1kRfzDlSnaY28d|1^zD`VuSP=rnccNDeY1)=eN4}iG)xP-Uln+kw zsibfo--2CQd^Y*n$x0kGLN8q(;N2Z|J&qvnme5u(dm6XFPE0*SDn%%tB+`tZGAiPJj!6|6=*~pq*C^h@_2%1@%lWBJ2x#rEyO0&C_W`pn^%?78S*&jyM>|&|e z??=$g(wtDUjpmJ;m1g&9%?9ATnhi`rv%ik4*tHu9l|Owa(CO9^RX6Q%Y~oAxe0UWc8DDQQ}g} z>omRx)YCShyuD1FMVYtPto23P>1iTsZ5~~-nl~48JwL$URACbIHB6J@_>-vWPKm2j zMLnhGyey^Gm|J$OT6)GpOhc=srzBUVYLn}8RXY#)aD?uRpX-TtSeT(xR+Zl8Te`8s zBjnBKN97p!3@t6ix(Nnn3pQvOMUOC?E{9nfVl=B_&!q;NsotOm2f5P2T%Yg6v5L?b z>zPEOZI*h8$cx+qs!dVrZ4L#YAaPOcq5`VKlFJ1IB`*VU_X8zd+vWqulA!GSLHU#? z=!Uog`G)MPvv?(&Q$*@<7Smy zKRGZnu*B~7PLGZ0xnkW~21|XZe`0`GFOUurhQjL$rA6xY&Z5G&li_&;i~AU#M$=bc zNI70nzU%V@{sA@toTH^Dyg2GWO;{xG-2~l_;2;+hdX8l(c59G69l)t({j8g~mE*y% z27tsD1#c7%vCO=Iz`Bv7?XN+J#!Ecq?ggl)8yB5-{XkN3 z3q?vPCv8B8t9k~?NQ6MqRfX4-Xl@5*py?LRB|LGfHWbM+Ll8>PUGi=12&EFawZ6W; zzrU6oD%4-s;uM`|d!(x807Q+h4HZ^=eJ z9QiDI{?!nHw;&HA#SUB%?1&H_xapZN#_6bMWEcGm_a_pS$Pj&-1u)cDLxy zEDfjeC({kWRfh=-hBm#yZW+v$!D?mVDT31iV#ZL5RYO{oZ1imJRvYfKUVxK~#D_!$ zgu>C{6>9cE3Op!}4ez$&u<{;*dT{_)VK@3#Y*?;_=zQP$I9XFgNBQF5TII}`p2lHe zDmrv3p7roj#l}Df{5oaw4pn*p{}dMbF#i!E2HiBoAbKbg=TyP4Iy3sfXDhJT@@b!r Vu`=T6R@DoguG_o0WUs8f_CJN2hyefq literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/.doctrees/textbooks.doctree b/doc/LectureNotes/_build/.doctrees/textbooks.doctree new file mode 100644 index 0000000000000000000000000000000000000000..7870083be2e5caf750ff2e988e01ffa47debcb10 GIT binary patch literal 23977 zcmeHPU5p%8R<;w5?eV|kwF%e`cf-Ou$@I_sc?!Wfy{jc<|MYF!K9e6#zCHv7KlxSJDWBStVzu%9}Q1Q@qf@&gvERe0x_S{t5 znz4>r$KLN>vu2~2&<-7$c@@pB$(^w2d7D9>3ZcvqdV(U^S0DA&mq68p=xEnK-C^H4 z-)kDt2_p=ByV(oz?ewiL+GarCb*$wWL4b$&{~6cl#Q%5P>~-~zrqSB;Wt+I6BJBzE^uUkbY3>SR2v}q+45)24Ha4YY)}m@*QdTF8u}`wRZqP z&zGwnzN!{-1FIdgZ(X)d^_pnqELvf#`$pHwqVaO$iEMctz;8-Zq>_zW3LHZ4xa@Q- zBd~+4*u}9@F`$>C32#$OCl$cAE?I}I7p>PYJ#2bR~J6OUn0(su=6y6)O!%ffI? zk#w}+o}Lp=yFPkc`g6hx!)|b^R@>g*t_I1|Dmt>Z-mQ^w*WSEwv%Yxq#^QYa#^OR% ztXXy-@NZ~I(J^c{?P^t1NN-zkxIp5sE!drCSw?n6s?rl znoO6KsurF9Ue=(<4d8zg?hI;SxU8}%Eh)F zTH9ud=R_ar=kam|)+iGb@y= zRr^uNiUG{uWr2Anr<7bbQ_iu3a#Rl~%K3$T8?Odz|GI{aC!c<}Mc*n)KP-^Y&$Da5 ztSJrNc=n>FAHPe{g_`(^=58|iP@EG7DME(AaH^tZxS}bAiO|6D44M_T5y-TC31VoA zU9TsMR!au)RHEnFG+VI4U6F8+*s=|16%QT@x7R_`LvakGd|^?zqXl9f!Hn%w_g$%p zSl>er|7S5TAe*Jx<7{HHqxjbw_oXX+#8gbosy75?dqF17GhD;j#dJHJf-k0lNM$Mo z%s5j@rLB*q6a-fPR~##vy`N67`k&YvhGy^9eP~p#iqG9yBcuO_vEz);6nSq|7Gx}w#O}l5~<->;#S(i61 z(O-p=CS;;#V%Rq>B$9cF7Ml97%(+y?nhfQg#}cUn@wV4)&xvKxw*7!Y3YA_!>fyGr zD+9tgYxW()7|s~{O3|f3tHq#IrUvb70@~{-0Fy^Vo-eYD1}&ez6QC`&V9np3P5EOu zF(htNI&%zRyuySrUV?2juUHCSlvfh=xH1*AL<8p&Xz!=cj$wigep{Rj61^#uQ;@;E z6}bgVec1E67UJ(=_-?_fANyX@Xxfzdvs>aZM!_!Sp|>#GfB0YxuQ5h4z-6h^9Z{A> zblApJ9YzH5Qqp1TX@}+e0?U}G%rgk&L20=~?tv0Drk$|yo79%e8^19%fEN+~KS;sI z2ash70VwYpTec~_SiMsv<6SXe#1C8qavda;5O5$A^!)f)oD7{)ksxQ9#j>;R`Iw@1 zz{q!uPSdwdjDqj#Xm7y0pHpob_&9XQUQ7V}^CbouNT%@NFMvL(#9%j zP_tdv+fpuHGmX%w1TEyJTCx^|J=4Zgp%CjBu>g#V{7wrYE`$Xw08vJXW^QrRhGewP zVWqPdp!u%-Nausj41RScgERaFr1{;%V0bvsBwoIhO}v!bUhe0EtRgjs=pbtWg#R%F z;p8y{Y2SnV7S7kaoY*t>txDP4Eycz(sO9JzV}Z{mA&PsredMpnly!=j@(&tM*H#wh z>-998)SiTy=D=hqo=vLEOV^jCtu{EQp+oI5u-jM?qJg}jxFJ~wPIw{qZNDZv$Trk4 z;JZdhe_6Gpez1ZmcPiiyEHrc+c3_=ncvAj{vf~|uYHb?_p&gbo`@%F+Y?{*iMB)iE ziKOy`!y-e8_yNuQehdrVCKlXZ&oby`sL~*pp$$#HS{PKYR~dCOOO5ZFYPN6S5~F>S z&kwBmg|ycvQ!?iWSmpUD2r>+wS76F)bi2NX z@}vB7r-W9r?U$lU8-FS4U|eq%wtg}cqLj-Cr7Wk$MOl&zrEFZGHUoB>1Wg*$5|p9c?-hm=gi=D4%qL?zX0=+bi)gUQ);z{y@|lj?{PU?X<})436k@u6 zrC_P@VD;W&y^;Scc?v$ zcGml}+)i6U)zVGr&Y)2|;EDk5u#M$&93!%unr}YGu}mS3`x``aOZA&|v1(xRHf`}U z-1VM;L@U-7Xo(%^!fAsw(iaD7@!W1`nS#Qxkr*wMEz7mP+@m~Dw~HN2*a+?k%o^24 zIs8%R{^X-ZY3$Yfmaq#JDXI;lNNM!$w+p)tlhf6Noc=oP-3jEx1;&um&rKIj4fPEJ zvQI4d0jNd5?BN%9()iLpm?TBBl9+2T`v<|Ea;5ckE`DDW={AFs6`DDW~g~;Xy zNkFxPfNEgpLYjo?BuS^Cn58%aagGUNFx4_LXI#eo&qO}H1%g2wH86qAX1~Uf6n05t z7ZRnm)ZS>debNx?o=1EZS`vRy%tfN9szA_rn|0sox#lQk-H0QvsqtX9-Lt$l!9A;h ziG9DYt1{!gk}%$XrzSRm@wmVk##2#8k<`JnZGkZ?OPz+aER8^T_Qj%EKMu!A0ESs{ zY-fE?6+c`b(m4OnKLs5$Y^V7w%c%ZW>H_&J%QA&n_U{qW-CJ6EvyT0hp5v|4$qi>0 zseqtsBabicc%3c|FLGR0AiW#s;FM16+(!40FsF3d6~n}7Em~b>5;vT{!&Xi#=$XRM z$w3I~vDMu!j6%lMnk!(m+ONf*vtK7yGR#V)SEjyqcKo5=0l!l6oTd`?#MI*6Y zwz`;%@C_{w{G*pH^#qc37SMhK1Zeb8_?%%P!nre^l#X)J&+ene?3Vk|p$%_eYG6=K zVK-$y6A7PLsXWK6`7riR*-3XDCb3ra>)9A>jR>FuY*FlB#819NSU9iv4(* z#r@Jr7x|cn?RqQV@R?pZ}N> zxMPA3=FvxcnCALfK?HW#Rft@jH%gHU)5E~-ycbbzc6*VADg9|-M`a%TY{G+olA6*4 z9^?XJcu>VX#YaEY@eI?RWHnBE5(iL!TNo&g>!xiV^Ve8_f9F$X_ZrYo+n1rQhAl6j z9~tvMPK`RBA6cddKfd=dwwW~6-a@pvR7WtFr7PkXqZfLv*YSFR2zCRkZ3e~De|8wu z63yz2ak_4V+XUz^1T8Oc`Y+uF5~u8F`zmfvQFqt`Dr6nQq^?x!5v~lkJ%2M8zcgix zPLM5cTadc+Jai_u3&@C823c}bCD|c`p zgjC_+NT4omknWa^^WgEN4wRs5?b5Nx?Bh}Xo|~qRn3ARw(ylX|91Q!D!X(6O{c6J2 z|CSo|1h(b^W7vABaMYTyM8;-s!WyH!$pfqZEsPZidvCQcV~+#*y*fO8z*m0cj`{AG z_WO_E+Y_5z-Wc}d{IQR%_q!=Y>De3@hy)dvK^-?Njjvwuye9vN;OKGR(dlt(S@+>d*tc7{8oNORT&v^4B zTP@A5;+{ue1w}~`qG{#D4m@#7`fk zT{~Et$=ldJWypwiLzQ3`I_P1x( zLwCLK*#UhXNy|iwbbc_XgkFV@K~|E(d=(?CAeU6pcd=BwcEzJ(kOp70nTS8@6r~oY zbbcb9xu5d{zA!oA$j3G_pEsM{HJYh(;Ng&}!?DHVc2ITI=VV0FJ5;SVs1pxIXIed< zE^uumcdM>Or}&Awc6Bv69>|cEPjMUV`Sg0(HYRN$?nUiKM}Nt1uscp)7vbejh)um?5$6q|?QZlW{e;E!5q_&lAbdp6}q->CkI6Xj_Wv9o!U4r!{x`(MxU5 z3te2_nt6%qU{BNCwzx;9v4h!phc1XcgSn~QZNzuBh5=Mwj-OypvYKKxDDBwb%V<$! zB9Popi^lsBJC<>qsoauIf8~Yfgk?DC&za~vrm(;YK~XQ@LNn2fDVx1@RCi`OzIPTS zkbdmo$!zEwElG{)e>FN|yIA3FU`MuR0#RJ5+Y8&3n-y%I?nhS!UpA!gdwzp1E_0+0 zoas|rVLv*HgRFyEW|WqGbam*>kcy3O!{wqgv0nWREPZA0#gK4xj@l3aC<_odXj4Br z&D$S=xQX8{;1c5Kf{&Y8eY0W8F5R2!VgovL#8G4p+Gr*avCRg~?X~TlzI8M@Ls^xE zeiXf6!>CT1y-v508XkaE4`_wnM`D^3o>fnPt{YJI>qp%{_Drv_(;<^OCwIDDARAdp zWhJe!v4zFH*hJ4ZOgk`|jzmL}uRO*MVZ9NZLSt}|1vY(RaS%#RC&e0V8+X2`dzZVi z+ej}@rEfV!PiO_NLAD(YWYN33xDUsz;_{rDDYt4U8PxjGMUo(mB+;O)^9~#b7d+#6 z<)`aQfaQ$eZVgpKPjod`c+^7P8<#-aGHk1A=qMwqm<^34T)+&i{>t&_Dh#UANPR2) z{6^5F?)_SHNqr4=_v3m_v>bLz7@08SlemLA{RziA7wqCryq$)Ha{xGAWx-63I38K} zsi1@OP~*!zx_udv7u4_gXT$NZ3nNDGD(=*Ur=Wjhw?dB*|1==I+3VtC=xS?qUv=W{ zUxSf&EgSbXX#anS7$N922g{w4xN6IUOYWjBOUe-1F1B!g*JcA-<7}sqPWthEFCz8S z$Mwq5@t6tVdaxp#g1{CP*i2~P#v?Eb?y)|PieMEXwyXdflWwm53b<X76zYe>*%ya(C2tNzC%_6AmC5J((Mi2tj)Cado#gq)0P&;#6SxMlkZ?A@ o9bow?t}rEsfyXD)$Et5nn8a6!Yt!%eP%AW zyE>{V>nP)JI5lwpo`X1?Lah9Jex;Ipuk~fAqx`TlYVTK39^rvevBx3;a4yH9Xkp<| zVIjvqjR}Z|3<;;28Oyh=p9V)o(ITx)OepUQjKd>>OrGUb=gSvaMcaQU5{FY+yZlkW z<>jx(;XaxH_w4eF%b)0HRQLta2j7YeEY6)N&FH3f?KvOj*#sPy&$+6ZFeIfAcp47h&9l%>Drmy>@n1jy?R6g_O(MPfs z*@tc5ExuaEZHf!czh4lh(>-5}-M^PO)!&~u*C#p^7}Ofy64_#(E?m5uUSu~ChA4vv zkJg{4B7F+#I5ZxHtxs~C51vk@Iw6~A7I)xLx-nH6rlxxjYy|~^B zB8~L2H$9|_g%iy!E#)z0z-T_q=bmNnG|~e%nD*)4+<7hUqL0Ex~LCc^fB+t ztXng`3&Z^llDa^4D&u%{b|^9BRPVGgc3Ed0U$K!pyuI2Nj?(#+D-_%9?rCIRZOqcrJDUjgYcx0C z4H>bGR^c2AcyVw_a@-?tYxqLhj>l~|Pa1nQ_WwsOyn~utj6^;xaih#$^qYI~7p&fQ z&;Qg09~Py}nk9mTIGOUirr-|IiY_&bw(|=n zMx>He_N)MB?V2Tb=q#;o?d-N3%p~=4THODYH9+^ z55tw;W(@bGkZ&Ms!FnRclB?H9DnsC(4<6wlLV{Q!FfcH?@-{F!zE7vdn|DE&q|8x? zBM4?kN3(@V2;`s;M(UxTGRy>AxL02n@`u19rxUc}Yy>#k%C3H7LTQrv{5Y8fXgn2#_UB1tQOYX}1 zx@_z;b|=26L47>`dL5JWq}>{9zY}*B`Ij2g61J3-m~D+Zg%5ZBp$h#YL%tT0cjIoq z#^H9pRm5M*_d{;)HO{{uV6Hi^d!LW5Zq0u>$-8pB^NH{bED3?#*mKUAN!z(E0kU(% zE<;9$uR{+zyvV&*A5{6O>6vsN_PTiPN{WUT?-VnLX~#;^5F9XE@-H(0^R8w*wBveV z1?c&n1ZFj!91f}DJJ~~rDFtkLP?-ss2Y=ZbxwH-&VR~22U1SE8jh$sD>vpv-s}+Zg zodsH}7LStb{o(uQ5vH?o&B@#sIj;)Owk$AyDUEfl(f4W9+)!h%a18T=`Yl_u$`L8F z!viuSSyFXJiuFVKNdbIWAm#}B5$aUushFn!Zg{5mH@YG1Do@UqaM@rdIZj_N_fxeW zT=~OZ$4O{z2CAw;*n}x1dk<(yaek}d1gnRNO|p}2usYa`QY7>iD;p@_IkiMC;RI&j zWz+9H9NL;Ra9UNS#a)Me95MZ&xk(-ts6J%iIxa;RfsXcm^hHZ-+~2V zsnDDbrtR-m)BqN{=kpRUmA#ZzH5o%VP39l+}nf_sYy#(5*(CuTNp$}PBZ#rI1C05q{^ z?>2?zjIe2d#H?j@md~Co3w5X<`v(S=dL`JVb$$^<+iz{Tbu=s&Nzjx-+9U_tL00!o zSxZ?^tWf)b!jMxvZSDn|=gFyp7WlziNKmH42&1Uh15!Cu)G?(W?Wg4IrbuTf;vl3c zqo#6C;Zk|l`T(CmGuk$8yz0KiSTCMIrmbRl!-jqzyS5OwAM|0=r!Th% z&R2-28#Yl?szY?h9f#>Ek#(LH^7{$Lqd(B^{b- zvDgdLfl!~%mw5@pwa+XTfI^%bzO#xV!gb;tQ$|umm+wA+OCG01u$##Whx}(+;>WaT zjmmP*^HI3%o>iirW@@B2s*5d5=?p*nLlpb}SiQ?@9~Q+s#NLb?7@^Z6`0^|l!B1BA z2?jQS*LVKW)|XXGc8%D=CF~CPQ^suRLa7qv~Q-dE3JkEdAjll`pvDd!xa9xINCBov?dgM zvkM885M2fo`tbS0q#Fp;+E<#Jo10fqu;CQRwsG&+U;_DD{hHfI;Bc>nFhf3snMVl& zO1i}Q@A&SJo%v)Zez%`bYkuv~E}%4E0H@1Bo+`*2ylce)pih7FfU}zPwAz|hVX^6i z##Qi)SQG1NTRe@W1$jB$(dz2;YRKRlt_^<9@l#ErRFSp^3P<~1o@ns8=2jDb^?ZYs zrOACa!>|sKapf^yy*mq$qB35B^xA&Y+P1n^HBI&fHVNZ7^t?JSUpXG%Wl>pQl?B43 zC)d1|>H9mv&Y6C=I+B%nN@oUOm$Ha!G16!B}dxa z{cw&;)msyJ`pV3AnnvBb&w`UK)4_?!kf9swo2ly(0dTWabZ6!!2C?9>qL8`k_hny+ zzvJ6vN_N4HJ>r;GzQ+$V0Rv^lzv}APPgWlIZEEUkK%ALWbw6k+&PN(@+u?%!{p(W8 zXn=bi_Os>M-|D5gO-oTXQ_fYPDRw@yPHBF1#cf1pOw$sC4|m6^2yUn3B8$mNM|B$# zeJDjwCs3P~&Amk3a%2PgnypUTQSqal5MOOXYwley5$BgD4!<>+5XK*gyhdLfT7Cdp zAs7a{(`#SqaSGmwE&>k9NhkW!HoTm;)Zn(HZUsSu#5(o)@DtRTm4f;Um6A%tCr1N2jHVRNE#V}(0^LE`)p(Mi%~o7kuPpjPDM**TCQiIX6l5bn z+ZMfG@*f;Gsd2lT_DejzjwjmBTPXvvfrVG#=st+Z*ltd3x9uQHs zS?n!jEowqwsOKvdmLYC#*LXknzBd!w!HJFR{GX<>y^`71WykdCiwrmnC;KS%Wv5R? zd$RO1K@oGr;MK!xzA;5TVuIizCX1l;d)9rMT_>iAQT${Wus9(^-(+4TSrB=yonJ1EECP^ zp)4MiTAjVM>L_STl*K^(JAT^7YBL=uo6+{`mN_6}PS}j9lE&3mJL20bw|zNnG1bO?%^Si7(m9V|NX z+;KVzeu0}<@H8Y330E8VEOd|EI+ry&S}8?;gysx0L;0q+N{)u(17X_-{-MM2;YGd5 z+VH|@aW-0MNo)E7TULu6YB*2`%dd%GA_=$X+xoa?%uXDZ2llCu)*PIJvZTemI~KQ8 z_@J*VSEkEB3Q|7uHxLM-eC+P*Mkm>IBu?TD$)Zqg2~sWn&Q*P?>(!De7h*9wH6|MJ zvil8~K!al4nz*Hpd+vQ~7=N2uhIo7PVld)*RrdkYg1{o4)1oIjMFH24YjnNnm^I(}&#r-G-JO7@7q^guG$+;)1B>{A#*PgI@fbu6im zgchy8RQk;WhP$ZxsoSjJh*}mUXkEQ~LEGS<2+w%tzxc~G8c3us-V};r&pbHaxi@n> xwVM2Gi8|{|h&DKuwfw5&!_oVGOw2MAGW8lQ$%mPF@^?ZwaIfp0@-O_q{s-fPUl{-Z literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/html/_panels_static/panels-main.c949a650a448cc0ae9fd3441c0e17fb0.css b/doc/LectureNotes/_build/html/_panels_static/panels-main.c949a650a448cc0ae9fd3441c0e17fb0.css new file mode 100644 index 000000000..fc14abc85 --- /dev/null +++ b/doc/LectureNotes/_build/html/_panels_static/panels-main.c949a650a448cc0ae9fd3441c0e17fb0.css @@ -0,0 +1 @@ +details.dropdown .summary-title{padding-right:3em !important;-moz-user-select:none;-ms-user-select:none;-webkit-user-select:none;user-select:none}details.dropdown:hover{cursor:pointer}details.dropdown .summary-content{cursor:default}details.dropdown summary{list-style:none;padding:1em}details.dropdown summary .octicon.no-title{vertical-align:middle}details.dropdown[open] summary .octicon.no-title{visibility:hidden}details.dropdown summary::-webkit-details-marker{display:none}details.dropdown summary:focus{outline:none}details.dropdown summary:hover .summary-up svg,details.dropdown summary:hover .summary-down svg{opacity:1}details.dropdown .summary-up svg,details.dropdown .summary-down svg{display:block;opacity:.6}details.dropdown .summary-up,details.dropdown .summary-down{pointer-events:none;position:absolute;right:1em;top:.75em}details.dropdown[open] .summary-down{visibility:hidden}details.dropdown:not([open]) .summary-up{visibility:hidden}details.dropdown.fade-in[open] summary~*{-moz-animation:panels-fade-in .5s ease-in-out;-webkit-animation:panels-fade-in .5s ease-in-out;animation:panels-fade-in .5s ease-in-out}details.dropdown.fade-in-slide-down[open] summary~*{-moz-animation:panels-fade-in .5s ease-in-out, panels-slide-down .5s ease-in-out;-webkit-animation:panels-fade-in .5s ease-in-out, panels-slide-down .5s ease-in-out;animation:panels-fade-in .5s ease-in-out, panels-slide-down .5s ease-in-out}@keyframes panels-fade-in{0%{opacity:0}100%{opacity:1}}@keyframes panels-slide-down{0%{transform:translate(0, -10px)}100%{transform:translate(0, 0)}}.octicon{display:inline-block;fill:currentColor;vertical-align:text-top}.tabbed-content{box-shadow:0 -.0625rem var(--tabs-color-overline),0 .0625rem var(--tabs-color-underline);display:none;order:99;padding-bottom:.75rem;padding-top:.75rem;width:100%}.tabbed-content>:first-child{margin-top:0 !important}.tabbed-content>:last-child{margin-bottom:0 !important}.tabbed-content>.tabbed-set{margin:0}.tabbed-set{border-radius:.125rem;display:flex;flex-wrap:wrap;margin:1em 0;position:relative}.tabbed-set>input{opacity:0;position:absolute}.tabbed-set>input:checked+label{border-color:var(--tabs-color-label-active);color:var(--tabs-color-label-active)}.tabbed-set>input:checked+label+.tabbed-content{display:block}.tabbed-set>input:focus+label{outline-style:auto}.tabbed-set>input:not(.focus-visible)+label{outline:none;-webkit-tap-highlight-color:transparent}.tabbed-set>label{border-bottom:.125rem solid transparent;color:var(--tabs-color-label-inactive);cursor:pointer;font-size:var(--tabs-size-label);font-weight:700;padding:1em 1.25em .5em;transition:color 250ms;width:auto;z-index:1}html .tabbed-set>label:hover{color:var(--tabs-color-label-active)} diff --git a/doc/LectureNotes/_build/html/_panels_static/panels-variables.06eb56fa6e07937060861dad626602ad.css b/doc/LectureNotes/_build/html/_panels_static/panels-variables.06eb56fa6e07937060861dad626602ad.css new file mode 100644 index 000000000..adc616622 --- /dev/null +++ b/doc/LectureNotes/_build/html/_panels_static/panels-variables.06eb56fa6e07937060861dad626602ad.css @@ -0,0 +1,7 @@ +:root { +--tabs-color-label-active: hsla(231, 99%, 66%, 1); +--tabs-color-label-inactive: rgba(178, 206, 245, 0.62); +--tabs-color-overline: rgb(207, 236, 238); +--tabs-color-underline: rgb(207, 236, 238); +--tabs-size-label: 1rem; +} \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/_sources/chapter1.ipynb b/doc/LectureNotes/_build/html/_sources/chapter1.ipynb new file mode 100644 index 000000000..7e2ad50e8 --- /dev/null +++ b/doc/LectureNotes/_build/html/_sources/chapter1.ipynb @@ -0,0 +1,4066 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Linear Regression, basic Elements\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK3155/h20/forelesningsvideoer/LectureAug21.mp4?vrtx=view-as-webpage)\n", + "\n", + "\n", + "## Introduction\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "Our emphasis throughout this series of lectures \n", + "is on understanding the mathematical aspects of\n", + "different algorithms used in the fields of data analysis and machine learning. \n", + "\n", + "However, where possible we will emphasize the\n", + "importance of using available software. We start thus with a hands-on\n", + "and top-down approach to machine learning. The aim is thus to start with\n", + "relevant data or data we have produced \n", + "and use these to introduce statistical data analysis\n", + "concepts and machine learning algorithms before we delve into the\n", + "algorithms themselves. The examples we will use in the beginning, start with simple\n", + "polynomials with random noise added. We will use the Python\n", + "software package [Scikit-Learn](http://scikit-learn.org/stable/) and\n", + "introduce various machine learning algorithms to make fits of\n", + "the data and predictions. We move thereafter to more interesting\n", + "cases such as data from say experiments (below we will look at experimental nuclear binding energies as an example).\n", + "These are examples where we can easily set up the data and\n", + "then use machine learning algorithms included in for example\n", + "**Scikit-Learn**. \n", + "\n", + "These examples will serve us the purpose of getting\n", + "started. Furthermore, they allow us to catch more than two birds with\n", + "a stone. They will allow us to bring in some programming specific\n", + "topics and tools as well as showing the power of various Python \n", + "libraries for machine learning and statistical data analysis. \n", + "\n", + "Here, we will mainly focus on two\n", + "specific Python packages for Machine Learning, Scikit-Learn and\n", + "Tensorflow (see below for links etc). Moreover, the examples we\n", + "introduce will serve as inputs to many of our discussions later, as\n", + "well as allowing you to set up models and produce your own data and\n", + "get started with programming.\n", + "\n", + "\n", + "\n", + "## What is Machine Learning?\n", + "\n", + "Statistics, data science and machine learning form important fields of\n", + "research in modern science. They describe how to learn and make\n", + "predictions from data, as well as allowing us to extract important\n", + "correlations about physical process and the underlying laws of motion\n", + "in large data sets. The latter, big data sets, appear frequently in\n", + "essentially all disciplines, from the traditional Science, Technology,\n", + "Mathematics and Engineering fields to Life Science, Law, education\n", + "research, the Humanities and the Social Sciences. \n", + "\n", + "It has become more\n", + "and more common to see research projects on big data in for example\n", + "the Social Sciences where extracting patterns from complicated survey\n", + "data is one of many research directions. Having a solid grasp of data\n", + "analysis and machine learning is thus becoming central to scientific\n", + "computing in many fields, and competences and skills within the fields\n", + "of machine learning and scientific computing are nowadays strongly\n", + "requested by many potential employers. The latter cannot be\n", + "overstated, familiarity with machine learning has almost become a\n", + "prerequisite for many of the most exciting employment opportunities,\n", + "whether they are in bioinformatics, life science, physics or finance,\n", + "in the private or the public sector. This author has had several\n", + "students or met students who have been hired recently based on their\n", + "skills and competences in scientific computing and data science, often\n", + "with marginal knowledge of machine learning.\n", + "\n", + "Machine learning is a subfield of computer science, and is closely\n", + "related to computational statistics. It evolved from the study of\n", + "pattern recognition in artificial intelligence (AI) research, and has\n", + "made contributions to AI tasks like computer vision, natural language\n", + "processing and speech recognition. Many of the methods we will study are also \n", + "strongly rooted in basic mathematics and physics research. \n", + "\n", + "Ideally, machine learning represents the science of giving computers\n", + "the ability to learn without being explicitly programmed. The idea is\n", + "that there exist generic algorithms which can be used to find patterns\n", + "in a broad class of data sets without having to write code\n", + "specifically for each problem. The algorithm will build its own logic\n", + "based on the data. You should however always keep in mind that\n", + "machines and algorithms are to a large extent developed by humans. The\n", + "insights and knowledge we have about a specific system, play a central\n", + "role when we develop a specific machine learning algorithm. \n", + "\n", + "Machine learning is an extremely rich field, in spite of its young\n", + "age. The increases we have seen during the last three decades in\n", + "computational capabilities have been followed by developments of\n", + "methods and techniques for analyzing and handling large date sets,\n", + "relying heavily on statistics, computer science and mathematics. The\n", + "field is rather new and developing rapidly. Popular software packages\n", + "written in Python for machine learning like\n", + "[Scikit-learn](http://scikit-learn.org/stable/),\n", + "[Tensorflow](https://www.tensorflow.org/),\n", + "[PyTorch](http://pytorch.org/) and [Keras](https://keras.io/), all\n", + "freely available at their respective GitHub sites, encompass\n", + "communities of developers in the thousands or more. And the number of\n", + "code developers and contributors keeps increasing. Not all the\n", + "algorithms and methods can be given a rigorous mathematical\n", + "justification, opening up thereby large rooms for experimenting and\n", + "trial and error and thereby exciting new developments. However, a\n", + "solid command of linear algebra, multivariate theory, probability\n", + "theory, statistical data analysis, understanding errors and Monte\n", + "Carlo methods are central elements in a proper understanding of many\n", + "of algorithms and methods we will discuss.\n", + "\n", + "\n", + "\n", + "The approaches to machine learning are many, but are often split into\n", + "two main categories. In *supervised learning* we know the answer to a\n", + "problem, and let the computer deduce the logic behind it. On the other\n", + "hand, *unsupervised learning* is a method for finding patterns and\n", + "relationship in data sets without any prior knowledge of the system.\n", + "Some authours also operate with a third category, namely\n", + "*reinforcement learning*. This is a paradigm of learning inspired by\n", + "behavioral psychology, where learning is achieved by trial-and-error,\n", + "solely from rewards and punishment.\n", + "\n", + "Another way to categorize machine learning tasks is to consider the\n", + "desired output of a system. Some of the most common tasks are:\n", + "\n", + " * Classification: Outputs are divided into two or more classes. The goal is to produce a model that assigns inputs into one of these classes. An example is to identify digits based on pictures of hand-written ones. Classification is typically supervised learning.\n", + "\n", + " * Regression: Finding a functional relationship between an input data set and a reference data set. The goal is to construct a function that maps input data to continuous output values.\n", + "\n", + " * Clustering: Data are divided into groups with certain common traits, without knowing the different groups beforehand. It is thus a form of unsupervised learning.\n", + "\n", + "The methods we cover have three main topics in common, irrespective of\n", + "whether we deal with supervised or unsupervised learning. The first\n", + "ingredient is normally our data set (which can be subdivided into\n", + "training and test data), the second item is a model which is normally a\n", + "function of some parameters. The model reflects our knowledge of the system (or lack thereof). As an example, if we know that our data show a behavior similar to what would be predicted by a polynomial, fitting our data to a polynomial of some degree would then determin our model. \n", + "\n", + "The last ingredient is a so-called **cost**\n", + "function which allows us to present an estimate on how good our model\n", + "is in reproducing the data it is supposed to train. \n", + "At the heart of basically all ML algorithms there are so-called minimization algorithms, often we end up with various variants of **gradient** methods.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Software and needed installations\n", + "\n", + "We will make extensive use of Python as programming language and its\n", + "myriad of available libraries. You will find\n", + "Jupyter notebooks invaluable in your work. You can run **R**\n", + "codes in the Jupyter/IPython notebooks, with the immediate benefit of\n", + "visualizing your data. You can also use compiled languages like C++,\n", + "Rust, Julia, Fortran etc if you prefer. The focus in these lectures will be\n", + "on Python.\n", + "\n", + "\n", + "If you have Python installed (we strongly recommend Python3) and you feel\n", + "pretty familiar with installing different packages, we recommend that\n", + "you install the following Python packages via **pip** as \n", + "\n", + "1. pip install numpy scipy matplotlib ipython scikit-learn mglearn sympy pandas pillow \n", + "\n", + "For Python3, replace **pip** with **pip3**.\n", + "\n", + "For OSX users we recommend, after having installed Xcode, to\n", + "install **brew**. Brew allows for a seamless installation of additional\n", + "software via for example \n", + "\n", + "1. brew install python3\n", + "\n", + "For Linux users, with its variety of distributions like for example the widely popular Ubuntu distribution,\n", + "you can use **pip** as well and simply install Python as \n", + "\n", + "1. sudo apt-get install python3 (or python for pyhton2.7)\n", + "\n", + "etc etc. \n", + "\n", + "\n", + "\n", + "## Python installers\n", + "\n", + "If you don't want to perform these operations separately and venture\n", + "into the hassle of exploring how to set up dependencies and paths, we\n", + "recommend two widely used distrubutions which set up all relevant\n", + "dependencies for Python, namely \n", + "\n", + "* [Anaconda](https://docs.anaconda.com/), \n", + "\n", + "which is an open source\n", + "distribution of the Python and R programming languages for large-scale\n", + "data processing, predictive analytics, and scientific computing, that\n", + "aims to simplify package management and deployment. Package versions\n", + "are managed by the package management system **conda**. \n", + "\n", + "* [Enthought canopy](https://www.enthought.com/product/canopy/) \n", + "\n", + "is a Python\n", + "distribution for scientific and analytic computing distribution and\n", + "analysis environment, available for free and under a commercial\n", + "license.\n", + "\n", + "Furthermore, [Google's Colab](https://colab.research.google.com/notebooks/welcome.ipynb) is a free Jupyter notebook environment that requires \n", + "no setup and runs entirely in the cloud. Try it out!\n", + "\n", + "\n", + "## Useful Python libraries\n", + "Here we list several useful Python libraries we strongly recommend (if you use anaconda many of these are already there)\n", + "\n", + "* [NumPy](https://www.numpy.org/) is a highly popular library for large, multi-dimensional arrays and matrices, along with a large collection of high-level mathematical functions to operate on these arrays\n", + "\n", + "* [The pandas](https://pandas.pydata.org/) library provides high-performance, easy-to-use data structures and data analysis tools \n", + "\n", + "* [Xarray](http://xarray.pydata.org/en/stable/) is a Python package that makes working with labelled multi-dimensional arrays simple, efficient, and fun!\n", + "\n", + "* [Scipy](https://www.scipy.org/) (pronounced “Sigh Pie”) is a Python-based ecosystem of open-source software for mathematics, science, and engineering. \n", + "\n", + "* [Matplotlib](https://matplotlib.org/) is a Python 2D plotting library which produces publication quality figures in a variety of hardcopy formats and interactive environments across platforms.\n", + "\n", + "* [Autograd](https://github.com/HIPS/autograd) can automatically differentiate native Python and Numpy code. It can handle a large subset of Python's features, including loops, ifs, recursion and closures, and it can even take derivatives of derivatives of derivatives\n", + "\n", + "* [SymPy](https://www.sympy.org/en/index.html) is a Python library for symbolic mathematics. \n", + "\n", + "* [scikit-learn](https://scikit-learn.org/stable/) has simple and efficient tools for machine learning, data mining and data analysis\n", + "\n", + "* [TensorFlow](https://www.tensorflow.org/) is a Python library for fast numerical computing created and released by Google\n", + "\n", + "* [Keras](https://keras.io/) is a high-level neural networks API, written in Python and capable of running on top of TensorFlow, CNTK, or Theano\n", + "\n", + "* And many more such as [pytorch](https://pytorch.org/), [Theano](https://pypi.org/project/Theano/) etc \n", + "\n", + "## Installing R, C++, cython or Julia\n", + "\n", + "You will also find it convenient to utilize **R**. We will mainly\n", + "use Python during our lectures and in various projects and exercises.\n", + "Those of you\n", + "already familiar with **R** should feel free to continue using **R**, keeping\n", + "however an eye on the parallel Python set ups. Similarly, if you are a\n", + "Python afecionado, feel free to explore **R** as well. Jupyter/Ipython\n", + "notebook allows you to run **R** codes interactively in your\n", + "browser. The software library **R** is really tailored for statistical data analysis\n", + "and allows for an easy usage of the tools and algorithms we will discuss in these\n", + "lectures.\n", + "\n", + "To install **R** with Jupyter notebook \n", + "[follow the link here](https://mpacer.org/maths/r-kernel-for-ipython-notebook)\n", + "\n", + "\n", + "\n", + "\n", + "## Installing R, C++, cython, Numba etc\n", + "\n", + "\n", + "For the C++ aficionados, Jupyter/IPython notebook allows you also to\n", + "install C++ and run codes written in this language interactively in\n", + "the browser. Since we will emphasize writing many of the algorithms\n", + "yourself, you can thus opt for either Python or C++ (or Fortran or other compiled languages) as programming\n", + "languages.\n", + "\n", + "To add more entropy, **cython** can also be used when running your\n", + "notebooks. It means that Python with the jupyter notebook\n", + "setup allows you to integrate widely popular softwares and tools for\n", + "scientific computing. Similarly, the \n", + "[Numba Python package](https://numba.pydata.org/) delivers increased performance\n", + "capabilities with minimal rewrites of your codes. With its\n", + "versatility, including symbolic operations, Python offers a unique\n", + "computational environment. Your jupyter notebook can easily be\n", + "converted into a nicely rendered **PDF** file or a Latex file for\n", + "further processing. For example, convert to latex as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + " pycod jupyter nbconvert filename.ipynb --to latex \n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "And to add more versatility, the Python package [SymPy](http://www.sympy.org/en/index.html) is a Python library for symbolic mathematics. It aims to become a full-featured computer algebra system (CAS) and is entirely written in Python. \n", + "\n", + "Finally, if you wish to use the light mark-up language \n", + "[doconce](https://github.com/hplgit/doconce) you can convert a standard ascii text file into various HTML \n", + "formats, ipython notebooks, latex files, pdf files etc with minimal edits. These lectures were generated using **doconce**.\n", + "\n", + "\n", + "\n", + "## Numpy examples and Important Matrix and vector handling packages\n", + "\n", + "There are several central software libraries for linear algebra and eigenvalue problems. Several of the more\n", + "popular ones have been wrapped into ofter software packages like those from the widely used text **Numerical Recipes**. The original source codes in many of the available packages are often taken from the widely used\n", + "software package LAPACK, which follows two other popular packages\n", + "developed in the 1970s, namely EISPACK and LINPACK. We describe them shortly here.\n", + "\n", + " * LINPACK: package for linear equations and least square problems.\n", + "\n", + " * LAPACK:package for solving symmetric, unsymmetric and generalized eigenvalue problems. From LAPACK's website it is possible to download for free all source codes from this library. Both C/C++ and Fortran versions are available.\n", + "\n", + " * BLAS (I, II and III): (Basic Linear Algebra Subprograms) are routines that provide standard building blocks for performing basic vector and matrix operations. Blas I is vector operations, II vector-matrix operations and III matrix-matrix operations. Highly parallelized and efficient codes, all available for download from .\n", + "\n", + "## Basic Matrix Features\n", + "\n", + "Matrix properties reminder" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{A} =\n", + " \\begin{bmatrix} a_{11} & a_{12} & a_{13} & a_{14} \\\\\n", + " a_{21} & a_{22} & a_{23} & a_{24} \\\\\n", + " a_{31} & a_{32} & a_{33} & a_{34} \\\\\n", + " a_{41} & a_{42} & a_{43} & a_{44}\n", + " \\end{bmatrix}\\qquad\n", + "\\mathbf{I} =\n", + " \\begin{bmatrix} 1 & 0 & 0 & 0 \\\\\n", + " 0 & 1 & 0 & 0 \\\\\n", + " 0 & 0 & 1 & 0 \\\\\n", + " 0 & 0 & 0 & 1\n", + " \\end{bmatrix}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The inverse of a matrix is defined by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{A}^{-1} \\cdot \\mathbf{A} = I\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "
Relations Name matrix elements
$A = A^{T}$ symmetric $a_{ij} = a_{ji}$
$A = \\left (A^{T} \\right )^{-1}$ real orthogonal $\\sum_k a_{ik} a_{jk} = \\sum_k a_{ki} a_{kj} = \\delta_{ij}$
$A = A^{ * }$ real matrix $a_{ij} = a_{ij}^{ * }$
$A = A^{\\dagger}$ hermitian $a_{ij} = a_{ji}^{ * }$
$A = \\left (A^{\\dagger} \\right )^{-1}$ unitary $\\sum_k a_{ik} a_{jk}^{ * } = \\sum_k a_{ki}^{ * } a_{kj} = \\delta_{ij}$
\n", + "\n", + "\n", + "### Some famous Matrices\n", + "\n", + " * Diagonal if $a_{ij}=0$ for $i\\ne j$\n", + "\n", + " * Upper triangular if $a_{ij}=0$ for $i > j$\n", + "\n", + " * Lower triangular if $a_{ij}=0$ for $i < j$\n", + "\n", + " * Upper Hessenberg if $a_{ij}=0$ for $i > j+1$\n", + "\n", + " * Lower Hessenberg if $a_{ij}=0$ for $i < j+1$\n", + "\n", + " * Tridiagonal if $a_{ij}=0$ for $|i -j| > 1$\n", + "\n", + " * Lower banded with bandwidth $p$: $a_{ij}=0$ for $i > j+p$\n", + "\n", + " * Upper banded with bandwidth $p$: $a_{ij}=0$ for $i < j+p$\n", + "\n", + " * Banded, block upper triangular, block lower triangular....\n", + "\n", + "### More Basic Matrix Features\n", + "\n", + "Some Equivalent Statements\n", + "For an $N\\times N$ matrix $\\mathbf{A}$ the following properties are all equivalent\n", + "\n", + " * If the inverse of $\\mathbf{A}$ exists, $\\mathbf{A}$ is nonsingular.\n", + "\n", + " * The equation $\\mathbf{Ax}=0$ implies $\\mathbf{x}=0$.\n", + "\n", + " * The rows of $\\mathbf{A}$ form a basis of $R^N$.\n", + "\n", + " * The columns of $\\mathbf{A}$ form a basis of $R^N$.\n", + "\n", + " * $\\mathbf{A}$ is a product of elementary matrices.\n", + "\n", + " * $0$ is not eigenvalue of $\\mathbf{A}$.\n", + "\n", + "## Numpy and arrays\n", + "[Numpy](http://www.numpy.org/) provides an easy way to handle arrays in Python. The standard way to import this library is as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here follows a simple example where we set up an array of ten elements, all determined by random numbers drawn according to the normal distribution," + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "n = 10\n", + "x = np.random.normal(size=n)\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We defined a vector $x$ with $n=10$ elements with its values given by the Normal distribution $N(0,1)$.\n", + "Another alternative is to declare a vector as follows" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "x = np.array([1, 2, 3])\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here we have defined a vector with three elements, with $x_0=1$, $x_1=2$ and $x_2=3$. Note that both Python and C++\n", + "start numbering array elements from $0$ and on. This means that a vector with $n$ elements has a sequence of entities $x_0, x_1, x_2, \\dots, x_{n-1}$. We could also let (recommended) Numpy to compute the logarithms of a specific array as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "x = np.log(np.array([4, 7, 8]))\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the last example we used Numpy's unary function $np.log$. This function is\n", + "highly tuned to compute array elements since the code is vectorized\n", + "and does not require looping. We normaly recommend that you use the\n", + "Numpy intrinsic functions instead of the corresponding **log** function\n", + "from Python's **math** module. The looping is done explicitely by the\n", + "**np.log** function. The alternative, and slower way to compute the\n", + "logarithms of a vector would be to write" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from math import log\n", + "x = np.array([4, 7, 8])\n", + "for i in range(0, len(x)):\n", + " x[i] = log(x[i])\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We note that our code is much longer already and we need to import the **log** function from the **math** module. \n", + "The attentive reader will also notice that the output is $[1, 1, 2]$. Python interprets automagically our numbers as integers (like the **automatic** keyword in C++). To change this we could define our array elements to be double precision numbers as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "x = np.log(np.array([4, 7, 8], dtype = np.float64))\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or simply write them as double precision numbers (Python uses 64 bits as default for floating point type variables), that is" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "x = np.log(np.array([4.0, 7.0, 8.0])\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "To check the number of bytes (remember that one byte contains eight bits for double precision variables), you can use simple use the **itemsize** functionality (the array $x$ is actually an object which inherits the functionalities defined in Numpy) as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "x = np.log(np.array([4.0, 7.0, 8.0])\n", + "print(x.itemsize)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Matrices in Python\n", + "\n", + "Having defined vectors, we are now ready to try out matrices. We can\n", + "define a $3 \\times 3 $ real matrix $\\hat{A}$ as (recall that we user\n", + "lowercase letters for vectors and uppercase letters for matrices)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))\n", + "print(A)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we use the **shape** function we would get $(3, 3)$ as output, that is verifying that our matrix is a $3\\times 3$ matrix. We can slice the matrix and print for example the first column (Python organized matrix elements in a row-major order, see below) as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))\n", + "# print the first column, row-major order and elements start with 0\n", + "print(A[:,0])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can continue this was by printing out other columns or rows. The example here prints out the second column" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))\n", + "# print the first column, row-major order and elements start with 0\n", + "print(A[1,:])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Numpy contains many other functionalities that allow us to slice, subdivide etc etc arrays. We strongly recommend that you look up the [Numpy website for more details](http://www.numpy.org/). Useful functions when defining a matrix are the **np.zeros** function which declares a matrix of a given dimension and sets all elements to zero" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "n = 10\n", + "# define a matrix of dimension 10 x 10 and set all elements to zero\n", + "A = np.zeros( (n, n) )\n", + "print(A)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or initializing all elements to" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "n = 10\n", + "# define a matrix of dimension 10 x 10 and set all elements to one\n", + "A = np.ones( (n, n) )\n", + "print(A)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or as unitarily distributed random numbers (see the material on random number generators in the statistics part)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "n = 10\n", + "# define a matrix of dimension 10 x 10 and set all elements to random numbers with x \\in [0, 1]\n", + "A = np.random.rand(n, n)\n", + "print(A)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "As we will see throughout these lectures, there are several extremely useful functionalities in Numpy.\n", + "As an example, consider the discussion of the covariance matrix. Suppose we have defined three vectors\n", + "$\\hat{x}, \\hat{y}, \\hat{z}$ with $n$ elements each. The covariance matrix is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\hat{\\Sigma} = \\begin{bmatrix} \\sigma_{xx} & \\sigma_{xy} & \\sigma_{xz} \\\\\n", + " \\sigma_{yx} & \\sigma_{yy} & \\sigma_{yz} \\\\\n", + " \\sigma_{zx} & \\sigma_{zy} & \\sigma_{zz} \n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where for example" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\sigma_{xy} =\\frac{1}{n} \\sum_{i=0}^{n-1}(x_i- \\overline{x})(y_i- \\overline{y}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The Numpy function **np.cov** calculates the covariance elements using the factor $1/(n-1)$ instead of $1/n$ since it assumes we do not have the exact mean values. \n", + "The following simple function uses the **np.vstack** function which takes each vector of dimension $1\\times n$ and produces a $3\\times n$ matrix $\\hat{W}$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\hat{W} = \\begin{bmatrix} x_0 & y_0 & z_0 \\\\\n", + " x_1 & y_1 & z_1 \\\\\n", + " x_2 & y_2 & z_2 \\\\\n", + " \\dots & \\dots & \\dots \\\\\n", + " x_{n-2} & y_{n-2} & z_{n-2} \\\\\n", + " x_{n-1} & y_{n-1} & z_{n-1}\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which in turn is converted into into the $3\\times 3$ covariance matrix\n", + "$\\hat{\\Sigma}$ via the Numpy function **np.cov()**. We note that we can also calculate\n", + "the mean value of each set of samples $\\hat{x}$ etc using the Numpy\n", + "function **np.mean(x)**. We can also extract the eigenvalues of the\n", + "covariance matrix through the **np.linalg.eig()** function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "import numpy as np\n", + "\n", + "n = 100\n", + "x = np.random.normal(size=n)\n", + "print(np.mean(x))\n", + "y = 4+3*x+np.random.normal(size=n)\n", + "print(np.mean(y))\n", + "z = x**3+np.random.normal(size=n)\n", + "print(np.mean(z))\n", + "W = np.vstack((x, y, z))\n", + "Sigma = np.cov(W)\n", + "print(Sigma)\n", + "Eigvals, Eigvecs = np.linalg.eig(Sigma)\n", + "print(Eigvals)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "%matplotlib inline\n", + "\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from scipy import sparse\n", + "eye = np.eye(4)\n", + "print(eye)\n", + "sparse_mtx = sparse.csr_matrix(eye)\n", + "print(sparse_mtx)\n", + "x = np.linspace(-10,10,100)\n", + "y = np.sin(x)\n", + "plt.plot(x,y,marker='x')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Meet the Pandas\n", + "\n", + "\n", + "\n", + "\n", + "Another useful Python package is\n", + "[pandas](https://pandas.pydata.org/), which is an open source library\n", + "providing high-performance, easy-to-use data structures and data\n", + "analysis tools for Python. **pandas** stands for panel data, a term borrowed from econometrics and is an efficient library for data analysis with an emphasis on tabular data.\n", + "**pandas** has two major classes, the **DataFrame** class with two-dimensional data objects and tabular data organized in columns and the class **Series** with a focus on one-dimensional data objects. Both classes allow you to index data easily as we will see in the examples below. \n", + "**pandas** allows you also to perform mathematical operations on the data, spanning from simple reshapings of vectors and matrices to statistical operations. \n", + "\n", + "The following simple example shows how we can, in an easy way make tables of our data. Here we define a data set which includes names, place of birth and date of birth, and displays the data in an easy to read way. We will see repeated use of **pandas**, in particular in connection with classification of data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import pandas as pd\n", + "from IPython.display import display\n", + "data = {'First Name': [\"Frodo\", \"Bilbo\", \"Aragorn II\", \"Samwise\"],\n", + " 'Last Name': [\"Baggins\", \"Baggins\",\"Elessar\",\"Gamgee\"],\n", + " 'Place of birth': [\"Shire\", \"Shire\", \"Eriador\", \"Shire\"],\n", + " 'Date of Birth T.A.': [2968, 2890, 2931, 2980]\n", + " }\n", + "data_pandas = pd.DataFrame(data)\n", + "display(data_pandas)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the above we have imported **pandas** with the shorthand **pd**, the latter has become the standard way we import **pandas**. We make then a list of various variables\n", + "and reorganize the aboves lists into a **DataFrame** and then print out a neat table with specific column labels as *Name*, *place of birth* and *date of birth*.\n", + "Displaying these results, we see that the indices are given by the default numbers from zero to three.\n", + "**pandas** is extremely flexible and we can easily change the above indices by defining a new type of indexing as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam'])\n", + "display(data_pandas)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Thereafter we display the content of the row which begins with the index **Aragorn**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "display(data_pandas.loc['Aragorn'])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can easily append data to this, for example" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "new_hobbit = {'First Name': [\"Peregrin\"],\n", + " 'Last Name': [\"Took\"],\n", + " 'Place of birth': [\"Shire\"],\n", + " 'Date of Birth T.A.': [2990]\n", + " }\n", + "data_pandas=data_pandas.append(pd.DataFrame(new_hobbit, index=['Pippin']))\n", + "display(data_pandas)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here are other examples where we use the **DataFrame** functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix \n", + "of dimensionality $10\\times 5$ and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import pandas as pd\n", + "from IPython.display import display\n", + "np.random.seed(100)\n", + "# setting up a 10 x 5 matrix\n", + "rows = 10\n", + "cols = 5\n", + "a = np.random.randn(rows,cols)\n", + "df = pd.DataFrame(a)\n", + "display(df)\n", + "print(df.mean())\n", + "print(df.std())\n", + "display(df**2)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Thereafter we can select specific columns only and plot final results" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']\n", + "df.index = np.arange(10)\n", + "\n", + "display(df)\n", + "print(df['Second'].mean() )\n", + "print(df.info())\n", + "print(df.describe())\n", + "\n", + "from pylab import plt, mpl\n", + "plt.style.use('seaborn')\n", + "mpl.rcParams['font.family'] = 'serif'\n", + "\n", + "df.cumsum().plot(lw=2.0, figsize=(10,6))\n", + "plt.show()\n", + "\n", + "\n", + "df.plot.bar(figsize=(10,6), rot=15)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can produce a $4\\times 4$ matrix" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "b = np.arange(16).reshape((4,4))\n", + "print(b)\n", + "df1 = pd.DataFrame(b)\n", + "print(df1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and many other operations. \n", + "\n", + "The **Series** class is another important class included in\n", + "**pandas**. You can view it as a specialization of **DataFrame** but where\n", + "we have just a single column of data. It shares many of the same features as _DataFrame. As with **DataFrame**,\n", + "most operations are vectorized, achieving thereby a high performance when dealing with computations of arrays, in particular labeled arrays.\n", + "As we will see below it leads also to a very concice code close to the mathematical operations we may be interested in.\n", + "For multidimensional arrays, we recommend strongly [xarray](http://xarray.pydata.org/en/stable/). **xarray** has much of the same flexibility as **pandas**, but allows for the extension to higher dimensions than two. We will see examples later of the usage of both **pandas** and **xarray**. \n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "In order to study various Machine Learning algorithms, we need to\n", + "access data. Acccessing data is an essential step in all machine\n", + "learning algorithms. In particular, setting up the so-called **design\n", + "matrix** (to be defined below) is often the first element we need in\n", + "order to perform our calculations. To set up the design matrix means\n", + "reading (and later, when the calculations are done, writing) data\n", + "in various formats, The formats span from reading files from disk,\n", + "loading data from databases and interacting with online sources\n", + "like web application programming interfaces (APIs).\n", + "\n", + "In handling various input formats, as discussed above, we will mainly stay with **pandas**,\n", + "a Python package which allows us, in a seamless and painless way, to\n", + "deal with a multitude of formats, from standard **csv** (comma separated\n", + "values) files, via **excel**, **html** to **hdf5** formats. With **pandas**\n", + "and the **DataFrame** and **Series** functionalities we are able to convert text data\n", + "into the calculational formats we need for a specific algorithm. And our code is going to be \n", + "pretty close the basic mathematical expressions.\n", + "\n", + "Our first data set is going to be a classic from nuclear physics, namely all\n", + "available data on binding energies. Don't be intimidated if you are not familiar with nuclear physics. It serves simply as an example here of a data set. \n", + "\n", + "We will show some of the\n", + "strengths of packages like **Scikit-Learn** in fitting nuclear binding energies to\n", + "specific functions using linear regression first. Then, as a teaser, we will show you how \n", + "you can easily implement other algorithms like decision trees and random forests and neural networks.\n", + "\n", + "But before we really start with nuclear physics data, let's just look at some simpler polynomial fitting cases, such as,\n", + "(don't be offended) fitting straight lines!\n", + "\n", + "\n", + "\n", + "\n", + "## Simple linear regression model using **scikit-learn**\n", + "\n", + "We start with perhaps our simplest possible example, using **Scikit-Learn** to perform linear regression analysis on a data set produced by us. \n", + "\n", + "What follows is a simple Python code where we have defined a function\n", + "$y$ in terms of the variable $x$. Both are defined as vectors with $100$ entries. \n", + "The numbers in the vector $\\hat{x}$ are given\n", + "by random numbers generated with a uniform distribution with entries\n", + "$x_i \\in [0,1]$ (more about probability distribution functions\n", + "later). These values are then used to define a function $y(x)$\n", + "(tabulated again as a vector) with a linear dependence on $x$ plus a\n", + "random noise added via the normal distribution.\n", + "\n", + "\n", + "The Numpy functions are imported used the **import numpy as np**\n", + "statement and the random number generator for the uniform distribution\n", + "is called using the function **np.random.rand()**, where we specificy\n", + "that we want $100$ random variables. Using Numpy we define\n", + "automatically an array with the specified number of elements, $100$ in\n", + "our case. With the Numpy function **randn()** we can compute random\n", + "numbers with the normal distribution (mean value $\\mu$ equal to zero and\n", + "variance $\\sigma^2$ set to one) and produce the values of $y$ assuming a linear\n", + "dependence as function of $x$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y = 2x+N(0,1),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $N(0,1)$ represents random numbers generated by the normal\n", + "distribution. From **Scikit-Learn** we import then the\n", + "**LinearRegression** functionality and make a prediction $\\tilde{y} =\n", + "\\alpha + \\beta x$ using the function **fit(x,y)**. We call the set of\n", + "data $(\\hat{x},\\hat{y})$ for our training data. The Python package\n", + "**scikit-learn** has also a functionality which extracts the above\n", + "fitting parameters $\\alpha$ and $\\beta$ (see below). Later we will\n", + "distinguish between training data and test data.\n", + "\n", + "For plotting we use the Python package\n", + "[matplotlib](https://matplotlib.org/) which produces publication\n", + "quality figures. Feel free to explore the extensive\n", + "[gallery](https://matplotlib.org/gallery/index.html) of examples. In\n", + "this example we plot our original values of $x$ and $y$ as well as the\n", + "prediction **ypredict** ($\\tilde{y}$), which attempts at fitting our\n", + "data with a straight line.\n", + "\n", + "The Python code follows here." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression\n", + "\n", + "x = np.random.rand(100,1)\n", + "y = 2*x+np.random.randn(100,1)\n", + "linreg = LinearRegression()\n", + "linreg.fit(x,y)\n", + "xnew = np.array([[0],[1]])\n", + "ypredict = linreg.predict(xnew)\n", + "\n", + "plt.plot(xnew, ypredict, \"r-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,1.0,0, 5.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Simple Linear Regression')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This example serves several aims. It allows us to demonstrate several\n", + "aspects of data analysis and later machine learning algorithms. The\n", + "immediate visualization shows that our linear fit is not\n", + "impressive. It goes through the data points, but there are many\n", + "outliers which are not reproduced by our linear regression. We could\n", + "now play around with this small program and change for example the\n", + "factor in front of $x$ and the normal distribution. Try to change the\n", + "function $y$ to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y = 10x+0.01 \\times N(0,1),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $x$ is defined as before. Does the fit look better? Indeed, by\n", + "reducing the role of the noise given by the normal distribution we see immediately that\n", + "our linear prediction seemingly reproduces better the training\n", + "set. However, this testing 'by the eye' is obviouly not satisfactory in the\n", + "long run. Here we have only defined the training data and our model, and \n", + "have not discussed a more rigorous approach to the **cost** function.\n", + "\n", + "We need more rigorous criteria in defining whether we have succeeded or\n", + "not in modeling our training data. You will be surprised to see that\n", + "many scientists seldomly venture beyond this 'by the eye' approach. A\n", + "standard approach for the *cost* function is the so-called $\\chi^2$\n", + "function (a variant of the mean-squared error (MSE))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\chi^2 = \\frac{1}{n}\n", + "\\sum_{i=0}^{n-1}\\frac{(y_i-\\tilde{y}_i)^2}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\sigma_i^2$ is the variance (to be defined later) of the entry\n", + "$y_i$. We may not know the explicit value of $\\sigma_i^2$, it serves\n", + "however the aim of scaling the equations and make the cost function\n", + "dimensionless. \n", + "\n", + "Minimizing the cost function is a central aspect of\n", + "our discussions to come. Finding its minima as function of the model\n", + "parameters ($\\alpha$ and $\\beta$ in our case) will be a recurring\n", + "theme in these series of lectures. Essentially all machine learning\n", + "algorithms we will discuss center around the minimization of the\n", + "chosen cost function. This depends in turn on our specific\n", + "model for describing the data, a typical situation in supervised\n", + "learning. Automatizing the search for the minima of the cost function is a\n", + "central ingredient in all algorithms. Typical methods which are\n", + "employed are various variants of **gradient** methods. These will be\n", + "discussed in more detail later. Again, you'll be surprised to hear that\n", + "many practitioners minimize the above function ''by the eye', popularly dubbed as \n", + "'chi by the eye'. That is, change a parameter and see (visually and numerically) that \n", + "the $\\chi^2$ function becomes smaller. \n", + "\n", + "There are many ways to define the cost function. A simpler approach is to look at the relative difference between the training data and the predicted data, that is we define \n", + "the relative error (why would we prefer the MSE instead of the relative error?) as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\epsilon_{\\mathrm{relative}}= \\frac{\\vert \\hat{y} -\\hat{\\tilde{y}}\\vert}{\\vert \\hat{y}\\vert}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The squared cost function results in an arithmetic mean-unbiased\n", + "estimator, and the absolute-value cost function results in a\n", + "median-unbiased estimator (in the one-dimensional case, and a\n", + "geometric median-unbiased estimator for the multi-dimensional\n", + "case). The squared cost function has the disadvantage that it has the tendency\n", + "to be dominated by outliers.\n", + "\n", + "We can modify easily the above Python code and plot the relative error instead" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression\n", + "\n", + "x = np.random.rand(100,1)\n", + "y = 5*x+0.01*np.random.randn(100,1)\n", + "linreg = LinearRegression()\n", + "linreg.fit(x,y)\n", + "ypredict = linreg.predict(x)\n", + "\n", + "plt.plot(x, np.abs(ypredict-y)/abs(y), \"ro\")\n", + "plt.axis([0,1.0,0.0, 0.5])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$\\epsilon_{\\mathrm{relative}}$')\n", + "plt.title(r'Relative error')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Depending on the parameter in front of the normal distribution, we may\n", + "have a small or larger relative error. Try to play around with\n", + "different training data sets and study (graphically) the value of the\n", + "relative error.\n", + "\n", + "As mentioned above, **Scikit-Learn** has an impressive functionality.\n", + "We can for example extract the values of $\\alpha$ and $\\beta$ and\n", + "their error estimates, or the variance and standard deviation and many\n", + "other properties from the statistical data analysis. \n", + "\n", + "Here we show an\n", + "example of the functionality of **Scikit-Learn**." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np \n", + "import matplotlib.pyplot as plt \n", + "from sklearn.linear_model import LinearRegression \n", + "from sklearn.metrics import mean_squared_error, r2_score, mean_squared_log_error, mean_absolute_error\n", + "\n", + "x = np.random.rand(100,1)\n", + "y = 2.0+ 5*x+0.5*np.random.randn(100,1)\n", + "linreg = LinearRegression()\n", + "linreg.fit(x,y)\n", + "ypredict = linreg.predict(x)\n", + "print('The intercept alpha: \\n', linreg.intercept_)\n", + "print('Coefficient beta : \\n', linreg.coef_)\n", + "# The mean squared error \n", + "print(\"Mean squared error: %.2f\" % mean_squared_error(y, ypredict))\n", + "# Explained variance score: 1 is perfect prediction \n", + "print('Variance score: %.2f' % r2_score(y, ypredict))\n", + "# Mean squared log error \n", + "print('Mean squared log error: %.2f' % mean_squared_log_error(y, ypredict) )\n", + "# Mean absolute error \n", + "print('Mean absolute error: %.2f' % mean_absolute_error(y, ypredict))\n", + "plt.plot(x, ypredict, \"r-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0.0,1.0,1.5, 7.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Linear Regression fit ')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The function **coef** gives us the parameter $\\beta$ of our fit while **intercept** yields \n", + "$\\alpha$. Depending on the constant in front of the normal distribution, we get values near or far from $alpha =2$ and $\\beta =5$. Try to play around with different parameters in front of the normal distribution. The function **meansquarederror** gives us the mean square error, a risk metric corresponding to the expected value of the squared (quadratic) error or loss defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "MSE(\\hat{y},\\hat{\\tilde{y}}) = \\frac{1}{n}\n", + "\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The smaller the value, the better the fit. Ideally we would like to\n", + "have an MSE equal zero. The attentive reader has probably recognized\n", + "this function as being similar to the $\\chi^2$ function defined above.\n", + "\n", + "The **r2score** function computes $R^2$, the coefficient of\n", + "determination. It provides a measure of how well future samples are\n", + "likely to be predicted by the model. Best possible score is 1.0 and it\n", + "can be negative (because the model can be arbitrarily worse). A\n", + "constant model that always predicts the expected value of $\\hat{y}$,\n", + "disregarding the input features, would get a $R^2$ score of $0.0$.\n", + "\n", + "If $\\tilde{\\hat{y}}_i$ is the predicted value of the $i-th$ sample and $y_i$ is the corresponding true value, then the score $R^2$ is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "R^2(\\hat{y}, \\tilde{\\hat{y}}) = 1 - \\frac{\\sum_{i=0}^{n - 1} (y_i - \\tilde{y}_i)^2}{\\sum_{i=0}^{n - 1} (y_i - \\bar{y})^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have defined the mean value of $\\hat{y}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\bar{y} = \\frac{1}{n} \\sum_{i=0}^{n - 1} y_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Another quantity taht we will meet again in our discussions of regression analysis is \n", + " the mean absolute error (MAE), a risk metric corresponding to the expected value of the absolute error loss or what we call the $l1$-norm loss. In our discussion above we presented the relative error.\n", + "The MAE is defined as follows" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\text{MAE}(\\hat{y}, \\hat{\\tilde{y}}) = \\frac{1}{n} \\sum_{i=0}^{n-1} \\left| y_i - \\tilde{y}_i \\right|.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We present the \n", + "squared logarithmic (quadratic) error" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\text{MSLE}(\\hat{y}, \\hat{\\tilde{y}}) = \\frac{1}{n} \\sum_{i=0}^{n - 1} (\\log_e (1 + y_i) - \\log_e (1 + \\tilde{y}_i) )^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\log_e (x)$ stands for the natural logarithm of $x$. This error\n", + "estimate is best to use when targets having exponential growth, such\n", + "as population counts, average sales of a commodity over a span of\n", + "years etc. \n", + "\n", + "\n", + "Finally, another cost function is the Huber cost function used in robust regression.\n", + "\n", + "The rationale behind this possible cost function is its reduced\n", + "sensitivity to outliers in the data set. In our discussions on\n", + "dimensionality reduction and normalization of data we will meet other\n", + "ways of dealing with outliers.\n", + "\n", + "The Huber cost function is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "H_{\\delta}(a)={\\begin{cases}{\\frac {1}{2}}{a^{2}}&{\\text{for }}|a|\\leq \\delta ,\\\\\\delta (|a|-{\\frac {1}{2}}\\delta ),&{\\text{otherwise.}}\\end{cases}}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here $a=\\boldsymbol{y} - \\boldsymbol{\\tilde{y}}$.\n", + "We will discuss in more\n", + "detail these and other functions in the various lectures. We conclude this part with another example. Instead of \n", + "a linear $x$-dependence we study now a cubic polynomial and use the polynomial regression analysis tools of scikit-learn." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "import random\n", + "from sklearn.linear_model import Ridge\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "from sklearn.pipeline import make_pipeline\n", + "from sklearn.linear_model import LinearRegression\n", + "\n", + "x=np.linspace(0.02,0.98,200)\n", + "noise = np.asarray(random.sample((range(200)),200))\n", + "y=x**3*noise\n", + "yn=x**3*100\n", + "poly3 = PolynomialFeatures(degree=3)\n", + "X = poly3.fit_transform(x[:,np.newaxis])\n", + "clf3 = LinearRegression()\n", + "clf3.fit(X,y)\n", + "\n", + "Xplot=poly3.fit_transform(x[:,np.newaxis])\n", + "poly3_plot=plt.plot(x, clf3.predict(Xplot), label='Cubic Fit')\n", + "plt.plot(x,yn, color='red', label=\"True Cubic\")\n", + "plt.scatter(x, y, label='Data', color='orange', s=15)\n", + "plt.legend()\n", + "plt.show()\n", + "\n", + "def error(a):\n", + " for i in y:\n", + " err=(y-yn)/yn\n", + " return abs(np.sum(err))/len(err)\n", + "\n", + "print (error(y))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Let us now dive into nuclear physics and remind ourselves briefly about some basic features about binding\n", + "energies. A basic quantity which can be measured for the ground\n", + "states of nuclei is the atomic mass $M(N, Z)$ of the neutral atom with\n", + "atomic mass number $A$ and charge $Z$. The number of neutrons is $N$. There are indeed several sophisticated experiments worldwide which allow us to measure this quantity to high precision (parts per million even). \n", + "\n", + "Atomic masses are usually tabulated in terms of the mass excess defined by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\Delta M(N, Z) = M(N, Z) - uA,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $u$ is the Atomic Mass Unit" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "u = M(^{12}\\mathrm{C})/12 = 931.4940954(57) \\hspace{0.1cm} \\mathrm{MeV}/c^2.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The nucleon masses are" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "m_p = 1.00727646693(9)u,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "m_n = 939.56536(8)\\hspace{0.1cm} \\mathrm{MeV}/c^2 = 1.0086649156(6)u.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the [2016 mass evaluation of by W.J.Huang, G.Audi, M.Wang, F.G.Kondev, S.Naimi and X.Xu](http://nuclearmasses.org/resources_folder/Wang_2017_Chinese_Phys_C_41_030003.pdf)\n", + "there are data on masses and decays of 3437 nuclei.\n", + "\n", + "The nuclear binding energy is defined as the energy required to break\n", + "up a given nucleus into its constituent parts of $N$ neutrons and $Z$\n", + "protons. In terms of the atomic masses $M(N, Z)$ the binding energy is\n", + "defined by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "BE(N, Z) = ZM_H c^2 + Nm_n c^2 - M(N, Z)c^2 ,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $M_H$ is the mass of the hydrogen atom and $m_n$ is the mass of the neutron.\n", + "In terms of the mass excess the binding energy is given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "BE(N, Z) = Z\\Delta_H c^2 + N\\Delta_n c^2 -\\Delta(N, Z)c^2 ,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\Delta_H c^2 = 7.2890$ MeV and $\\Delta_n c^2 = 8.0713$ MeV.\n", + "\n", + "\n", + "A popular and physically intuitive model which can be used to parametrize \n", + "the experimental binding energies as function of $A$, is the so-called \n", + "**liquid drop model**. The ansatz is based on the following expression" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "BE(N,Z) = a_1A-a_2A^{2/3}-a_3\\frac{Z^2}{A^{1/3}}-a_4\\frac{(N-Z)^2}{A},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $A$ stands for the number of nucleons and the $a_i$s are parameters which are determined by a fit \n", + "to the experimental data. \n", + "\n", + "\n", + "\n", + "\n", + "To arrive at the above expression we have assumed that we can make the following assumptions:\n", + "\n", + " * There is a volume term $a_1A$ proportional with the number of nucleons (the energy is also an extensive quantity). When an assembly of nucleons of the same size is packed together into the smallest volume, each interior nucleon has a certain number of other nucleons in contact with it. This contribution is proportional to the volume.\n", + "\n", + " * There is a surface energy term $a_2A^{2/3}$. The assumption here is that a nucleon at the surface of a nucleus interacts with fewer other nucleons than one in the interior of the nucleus and hence its binding energy is less. This surface energy term takes that into account and is therefore negative and is proportional to the surface area.\n", + "\n", + " * There is a Coulomb energy term $a_3\\frac{Z^2}{A^{1/3}}$. The electric repulsion between each pair of protons in a nucleus yields less binding. \n", + "\n", + " * There is an asymmetry term $a_4\\frac{(N-Z)^2}{A}$. This term is associated with the Pauli exclusion principle and reflects the fact that the proton-neutron interaction is more attractive on the average than the neutron-neutron and proton-proton interactions.\n", + "\n", + "We could also add a so-called pairing term, which is a correction term that\n", + "arises from the tendency of proton pairs and neutron pairs to\n", + "occur. An even number of particles is more stable than an odd number. \n", + "\n", + "\n", + "### Organizing our data\n", + "\n", + "Let us start with reading and organizing our data. \n", + "We start with the compilation of masses and binding energies from 2016.\n", + "After having downloaded this file to our own computer, we are now ready to read the file and start structuring our data.\n", + "\n", + "\n", + "We start with preparing folders for storing our calculations and the data file over masses and binding energies. We import also various modules that we will find useful in order to present various Machine Learning methods. Here we focus mainly on the functionality of **scikit-learn**." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "import sklearn.linear_model as skl\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\n", + "import os\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"MassEval2016.dat\"),'r')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Before we proceed, we define also a function for making our plots. You can obviously avoid this and simply set up various **matplotlib** commands every time you need them. You may however find it convenient to collect all such commands in one function and simply call this function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from pylab import plt, mpl\n", + "plt.style.use('seaborn')\n", + "mpl.rcParams['font.family'] = 'serif'\n", + "\n", + "def MakePlot(x,y, styles, labels, axlabels):\n", + " plt.figure(figsize=(10,6))\n", + " for i in range(len(x)):\n", + " plt.plot(x[i], y[i], styles[i], label = labels[i])\n", + " plt.xlabel(axlabels[0])\n", + " plt.ylabel(axlabels[1])\n", + " plt.legend(loc=0)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Our next step is to read the data on experimental binding energies and\n", + "reorganize them as functions of the mass number $A$, the number of\n", + "protons $Z$ and neutrons $N$ using **pandas**. Before we do this it is\n", + "always useful (unless you have a binary file or other types of compressed\n", + "data) to actually open the file and simply take a look at it!\n", + "\n", + "\n", + "In particular, the program that outputs the final nuclear masses is written in Fortran with a specific format. It means that we need to figure out the format and which columns contain the data we are interested in. Pandas comes with a function that reads formatted output. After having admired the file, we are now ready to start massaging it with **pandas**. The file begins with some basic format information." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\"\"\" \n", + "This is taken from the data file of the mass 2016 evaluation. \n", + "All files are 3436 lines long with 124 character per line. \n", + " Headers are 39 lines long. \n", + " col 1 : Fortran character control: 1 = page feed 0 = line feed \n", + " format : a1,i3,i5,i5,i5,1x,a3,a4,1x,f13.5,f11.5,f11.3,f9.3,1x,a2,f11.3,f9.3,1x,i3,1x,f12.5,f11.5 \n", + " These formats are reflected in the pandas widths variable below, see the statement \n", + " widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1), \n", + " Pandas has also a variable header, with length 39 in this case. \n", + "\"\"\"" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The data we are interested in are in columns 2, 3, 4 and 11, giving us\n", + "the number of neutrons, protons, mass numbers and binding energies,\n", + "respectively. We add also for the sake of completeness the element name. The data are in fixed-width formatted lines and we will\n", + "covert them into the **pandas** DataFrame structure." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Read the experimental data with Pandas\n", + "Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11),\n", + " names=('N', 'Z', 'A', 'Element', 'Ebinding'),\n", + " widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),\n", + " header=39,\n", + " index_col=False)\n", + "\n", + "# Extrapolated values are indicated by '#' in place of the decimal place, so\n", + "# the Ebinding column won't be numeric. Coerce to float and drop these entries.\n", + "Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce')\n", + "Masses = Masses.dropna()\n", + "# Convert from keV to MeV.\n", + "Masses['Ebinding'] /= 1000\n", + "\n", + "# Group the DataFrame by nucleon number, A.\n", + "Masses = Masses.groupby('A')\n", + "# Find the rows of the grouped DataFrame with the maximum binding energy.\n", + "Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We have now read in the data, grouped them according to the variables we are interested in. \n", + "We see how easy it is to reorganize the data using **pandas**. If we\n", + "were to do these operations in C/C++ or Fortran, we would have had to\n", + "write various functions/subroutines which perform the above\n", + "reorganizations for us. Having reorganized the data, we can now start\n", + "to make some simple fits using both the functionalities in **numpy** and\n", + "**Scikit-Learn** afterwards. \n", + "\n", + "Now we define five variables which contain\n", + "the number of nucleons $A$, the number of protons $Z$ and the number of neutrons $N$, the element name and finally the energies themselves." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "A = Masses['A']\n", + "Z = Masses['Z']\n", + "N = Masses['N']\n", + "Element = Masses['Element']\n", + "Energies = Masses['Ebinding']\n", + "print(Masses)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The next step, and we will define this mathematically later, is to set up the so-called **design matrix**. We will throughout call this matrix $\\boldsymbol{X}$.\n", + "It has dimensionality $p\\times n$, where $n$ is the number of data points and $p$ are the so-called predictors. In our case here they are given by the number of polynomials in $A$ we wish to include in the fit." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Now we set up the design matrix X\n", + "X = np.zeros((len(A),5))\n", + "X[:,0] = 1\n", + "X[:,1] = A\n", + "X[:,2] = A**(2.0/3.0)\n", + "X[:,3] = A**(-1.0/3.0)\n", + "X[:,4] = A**(-1.0)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With **scikitlearn** we are now ready to use linear regression and fit our data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "clf = skl.LinearRegression().fit(X, Energies)\n", + "fity = clf.predict(X)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Pretty simple! \n", + "Now we can print measures of how our fit is doing, the coefficients from the fits and plot the final fit together with our data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# The mean squared error \n", + "print(\"Mean squared error: %.2f\" % mean_squared_error(Energies, fity))\n", + "# Explained variance score: 1 is perfect prediction \n", + "print('Variance score: %.2f' % r2_score(Energies, fity))\n", + "# Mean absolute error \n", + "print('Mean absolute error: %.2f' % mean_absolute_error(Energies, fity))\n", + "print(clf.coef_, clf.intercept_)\n", + "\n", + "Masses['Eapprox'] = fity\n", + "# Generate a plot comparing the experimental with the fitted values values.\n", + "fig, ax = plt.subplots()\n", + "ax.set_xlabel(r'$A = N + Z$')\n", + "ax.set_ylabel(r'$E_\\mathrm{bind}\\,/\\mathrm{MeV}$')\n", + "ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,\n", + " label='Ame2016')\n", + "ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',\n", + " label='Fit')\n", + "ax.legend()\n", + "save_fig(\"Masses2016\")\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "As a teaser, let us now see how we can do this with decision trees using **scikit-learn**. Later we will switch to so-called **random forests**!" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\n", + "#Decision Tree Regression\n", + "from sklearn.tree import DecisionTreeRegressor\n", + "regr_1=DecisionTreeRegressor(max_depth=5)\n", + "regr_2=DecisionTreeRegressor(max_depth=7)\n", + "regr_3=DecisionTreeRegressor(max_depth=9)\n", + "regr_1.fit(X, Energies)\n", + "regr_2.fit(X, Energies)\n", + "regr_3.fit(X, Energies)\n", + "\n", + "\n", + "y_1 = regr_1.predict(X)\n", + "y_2 = regr_2.predict(X)\n", + "y_3=regr_3.predict(X)\n", + "Masses['Eapprox'] = y_3\n", + "# Plot the results\n", + "plt.figure()\n", + "plt.plot(A, Energies, color=\"blue\", label=\"Data\", linewidth=2)\n", + "plt.plot(A, y_1, color=\"red\", label=\"max_depth=5\", linewidth=2)\n", + "plt.plot(A, y_2, color=\"green\", label=\"max_depth=7\", linewidth=2)\n", + "plt.plot(A, y_3, color=\"m\", label=\"max_depth=9\", linewidth=2)\n", + "\n", + "plt.xlabel(\"$A$\")\n", + "plt.ylabel(\"$E$[MeV]\")\n", + "plt.title(\"Decision Tree Regression\")\n", + "plt.legend()\n", + "save_fig(\"Masses2016Trees\")\n", + "plt.show()\n", + "print(Masses)\n", + "print(np.mean( (Energies-y_1)**2))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The **seaborn** package allows us to visualize data in an efficient way. Note that we use **scikit-learn**'s multi-layer perceptron (or feed forward neural network) \n", + "functionality." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from sklearn.neural_network import MLPRegressor\n", + "from sklearn.metrics import accuracy_score\n", + "import seaborn as sns\n", + "\n", + "X_train = X\n", + "Y_train = Energies\n", + "n_hidden_neurons = 100\n", + "epochs = 100\n", + "# store models for later use\n", + "eta_vals = np.logspace(-5, 1, 7)\n", + "lmbd_vals = np.logspace(-5, 1, 7)\n", + "# store the models for later use\n", + "DNN_scikit = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)\n", + "train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))\n", + "sns.set()\n", + "for i, eta in enumerate(eta_vals):\n", + " for j, lmbd in enumerate(lmbd_vals):\n", + " dnn = MLPRegressor(hidden_layer_sizes=(n_hidden_neurons), activation='logistic',\n", + " alpha=lmbd, learning_rate_init=eta, max_iter=epochs)\n", + " dnn.fit(X_train, Y_train)\n", + " DNN_scikit[i][j] = dnn\n", + " train_accuracy[i][j] = dnn.score(X_train, Y_train)\n", + "\n", + "fig, ax = plt.subplots(figsize = (10, 10))\n", + "sns.heatmap(train_accuracy, annot=True, ax=ax, cmap=\"viridis\")\n", + "ax.set_title(\"Training Accuracy\")\n", + "ax.set_ylabel(\"$\\eta$\")\n", + "ax.set_xlabel(\"$\\lambda$\")\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Linear Regression, basic elements\n", + "\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureAug27.mp4?vrtx=view-as-webpage).\n", + "\n", + "\n", + "Fitting a continuous function with linear parameterization in terms of the parameters $\\boldsymbol{\\beta}$.\n", + "* Method of choice for fitting a continuous function!\n", + "\n", + "* Gives an excellent introduction to central Machine Learning features with **understandable pedagogical** links to other methods like **Neural Networks**, **Support Vector Machines** etc\n", + "\n", + "* Analytical expression for the fitting parameters $\\boldsymbol{\\beta}$\n", + "\n", + "* Analytical expressions for statistical propertiers like mean values, variances, confidence intervals and more\n", + "\n", + "* Analytical relation with probabilistic interpretations \n", + "\n", + "* Easy to introduce basic concepts like bias-variance tradeoff, cross-validation, resampling and regularization techniques and many other ML topics\n", + "\n", + "* Easy to code! And links well with classification problems and logistic regression and neural networks\n", + "\n", + "* Allows for **easy** hands-on understanding of gradient descent methods\n", + "\n", + "* and many more features\n", + "\n", + "For more discussions of Ridge and Lasso regression, [Wessel van Wieringen's](https://arxiv.org/abs/1509.09169) article is highly recommended.\n", + "Similarly, [Mehta et al's article](https://arxiv.org/abs/1803.08823) is also recommended.\n", + "\n", + "\n", + "\n", + "Regression modeling deals with the description of the sampling distribution of a given random variable $y$ and how it varies as function of another variable or a set of such variables $\\boldsymbol{x} =[x_0, x_1,\\dots, x_{n-1}]^T$. \n", + "The first variable is called the **dependent**, the **outcome** or the **response** variable while the set of variables $\\boldsymbol{x}$ is called the independent variable, or the predictor variable or the explanatory variable. \n", + "\n", + "A regression model aims at finding a likelihood function $p(\\boldsymbol{y}\\vert \\boldsymbol{x})$, that is the conditional distribution for $\\boldsymbol{y}$ with a given $\\boldsymbol{x}$. The estimation of $p(\\boldsymbol{y}\\vert \\boldsymbol{x})$ is made using a data set with \n", + "* $n$ cases $i = 0, 1, 2, \\dots, n-1$ \n", + "\n", + "* Response (target, dependent or outcome) variable $y_i$ with $i = 0, 1, 2, \\dots, n-1$ \n", + "\n", + "* $p$ so-called explanatory (independent or predictor) variables $\\boldsymbol{x}_i=[x_{i0}, x_{i1}, \\dots, x_{ip-1}]$ with $i = 0, 1, 2, \\dots, n-1$ and explanatory variables running from $0$ to $p-1$. See below for more explicit examples. \n", + "\n", + " The goal of the regression analysis is to extract/exploit relationship between $\\boldsymbol{y}$ and $\\boldsymbol{x}$ in or to infer causal dependencies, approximations to the likelihood functions, functional relationships and to make predictions, making fits and many other things.\n", + "\n", + "\n", + "Consider an experiment in which $p$ characteristics of $n$ samples are\n", + "measured. The data from this experiment, for various explanatory variables $p$ are normally represented by a matrix \n", + "$\\mathbf{X}$.\n", + "\n", + "The matrix $\\mathbf{X}$ is called the *design\n", + "matrix*. Additional information of the samples is available in the\n", + "form of $\\boldsymbol{y}$ (also as above). The variable $\\boldsymbol{y}$ is\n", + "generally referred to as the *response variable*. The aim of\n", + "regression analysis is to explain $\\boldsymbol{y}$ in terms of\n", + "$\\boldsymbol{X}$ through a functional relationship like $y_i =\n", + "f(\\mathbf{X}_{i,\\ast})$. When no prior knowledge on the form of\n", + "$f(\\cdot)$ is available, it is common to assume a linear relationship\n", + "between $\\boldsymbol{X}$ and $\\boldsymbol{y}$. This assumption gives rise to\n", + "the *linear regression model* where $\\boldsymbol{\\beta} = [\\beta_0, \\ldots,\n", + "\\beta_{p-1}]^{T}$ are the *regression parameters*. \n", + "\n", + "Linear regression gives us a set of analytical equations for the parameters $\\beta_j$.\n", + "\n", + "\n", + "In order to understand the relation among the predictors $p$, the set of data $n$ and the target (outcome, output etc) $\\boldsymbol{y}$,\n", + "consider the model we discussed for describing nuclear binding energies. \n", + "\n", + "There we assumed that we could parametrize the data using a polynomial approximation based on the liquid drop model.\n", + "Assuming" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "BE(A) = a_0+a_1A+a_2A^{2/3}+a_3A^{-1/3}+a_4A^{-1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we have five predictors, that is the intercept, the $A$ dependent term, the $A^{2/3}$ term and the $A^{-1/3}$ and $A^{-1}$ terms.\n", + "This gives $p=0,1,2,3,4$. Furthermore we have $n$ entries for each predictor. It means that our design matrix is a \n", + "$p\\times n$ matrix $\\boldsymbol{X}$.\n", + "\n", + "Here the predictors are based on a model we have made. A popular data set which is widely encountered in ML applications is the\n", + "so-called [credit card default data from Taiwan](https://www.sciencedirect.com/science/article/pii/S0957417407006719?via%3Dihub). The data set contains data on $n=30000$ credit card holders with predictors like gender, marital status, age, profession, education, etc. In total there are $24$ such predictors or attributes leading to a design matrix of dimensionality $24 \\times 30000$. This is however a classification problem and we will come back to it when we discuss Logistic Regression. \n", + "\n", + "\n", + "Before we proceed let us study a case from linear algebra where we aim at fitting a set of data $\\boldsymbol{y}=[y_0,y_1,\\dots,y_{n-1}]$. We could think of these data as a result of an experiment or a complicated numerical experiment. These data are functions of a series of variables $\\boldsymbol{x}=[x_0,x_1,\\dots,x_{n-1}]$, that is $y_i = y(x_i)$ with $i=0,1,2,\\dots,n-1$. The variables $x_i$ could represent physical quantities like time, temperature, position etc. We assume that $y(x)$ is a smooth function. \n", + "\n", + "Since obtaining these data points may not be trivial, we want to use these data to fit a function which can allow us to make predictions for values of $y$ which are not in the present set. The perhaps simplest approach is to assume we can parametrize our function in terms of a polynomial of degree $n-1$ with $n$ points, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y=y(x) \\rightarrow y(x_i)=\\tilde{y}_i+\\epsilon_i=\\sum_{j=0}^{n-1} \\beta_j x_i^j+\\epsilon_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\epsilon_i$ is the error in our approximation. \n", + "\n", + "\n", + "For every set of values $y_i,x_i$ we have thus the corresponding set of equations" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "y_0&=\\beta_0+\\beta_1x_0^1+\\beta_2x_0^2+\\dots+\\beta_{n-1}x_0^{n-1}+\\epsilon_0\\\\\n", + "y_1&=\\beta_0+\\beta_1x_1^1+\\beta_2x_1^2+\\dots+\\beta_{n-1}x_1^{n-1}+\\epsilon_1\\\\\n", + "y_2&=\\beta_0+\\beta_1x_2^1+\\beta_2x_2^2+\\dots+\\beta_{n-1}x_2^{n-1}+\\epsilon_2\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{n-1}&=\\beta_0+\\beta_1x_{n-1}^1+\\beta_2x_{n-1}^2+\\dots+\\beta_{n-1}x_{n-1}^{n-1}+\\epsilon_{n-1}.\\\\\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Defining the vectors" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y} = [y_0,y_1, y_2,\\dots, y_{n-1}]^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta} = [\\beta_0,\\beta_1, \\beta_2,\\dots, \\beta_{n-1}]^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\epsilon} = [\\epsilon_0,\\epsilon_1, \\epsilon_2,\\dots, \\epsilon_{n-1}]^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and the design matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\n", + "\\begin{bmatrix} \n", + "1& x_{0}^1 &x_{0}^2& \\dots & \\dots &x_{0}^{n-1}\\\\\n", + "1& x_{1}^1 &x_{1}^2& \\dots & \\dots &x_{1}^{n-1}\\\\\n", + "1& x_{2}^1 &x_{2}^2& \\dots & \\dots &x_{2}^{n-1}\\\\ \n", + "\\dots& \\dots &\\dots& \\dots & \\dots &\\dots\\\\\n", + "1& x_{n-1}^1 &x_{n-1}^2& \\dots & \\dots &x_{n-1}^{n-1}\\\\\n", + "\\end{bmatrix}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we can rewrite our equations as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y} = \\boldsymbol{X}\\boldsymbol{\\beta}+\\boldsymbol{\\epsilon}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The above design matrix is called a [Vandermonde matrix](https://en.wikipedia.org/wiki/Vandermonde_matrix).\n", + "\n", + "We are obviously not limited to the above polynomial expansions. We\n", + "could replace the various powers of $x$ with elements of Fourier\n", + "series or instead of $x_i^j$ we could have $\\cos{(j x_i)}$ or $\\sin{(j\n", + "x_i)}$, or time series or other orthogonal functions. For every set\n", + "of values $y_i,x_i$ we can then generalize the equations to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "y_0&=\\beta_0x_{00}+\\beta_1x_{01}+\\beta_2x_{02}+\\dots+\\beta_{n-1}x_{0n-1}+\\epsilon_0\\\\\n", + "y_1&=\\beta_0x_{10}+\\beta_1x_{11}+\\beta_2x_{12}+\\dots+\\beta_{n-1}x_{1n-1}+\\epsilon_1\\\\\n", + "y_2&=\\beta_0x_{20}+\\beta_1x_{21}+\\beta_2x_{22}+\\dots+\\beta_{n-1}x_{2n-1}+\\epsilon_2\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{i}&=\\beta_0x_{i0}+\\beta_1x_{i1}+\\beta_2x_{i2}+\\dots+\\beta_{n-1}x_{in-1}+\\epsilon_i\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{n-1}&=\\beta_0x_{n-1,0}+\\beta_1x_{n-1,2}+\\beta_2x_{n-1,2}+\\dots+\\beta_{n-1}x_{n-1,n-1}+\\epsilon_{n-1}.\\\\\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "**Note that we have $p=n$ here. The matrix is symmetric. This is generally not the case!**\n", + "\n", + "We redefine in turn the matrix $\\boldsymbol{X}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\n", + "\\begin{bmatrix} \n", + "x_{00}& x_{01} &x_{02}& \\dots & \\dots &x_{0,n-1}\\\\\n", + "x_{10}& x_{11} &x_{12}& \\dots & \\dots &x_{1,n-1}\\\\\n", + "x_{20}& x_{21} &x_{22}& \\dots & \\dots &x_{2,n-1}\\\\ \n", + "\\dots& \\dots &\\dots& \\dots & \\dots &\\dots\\\\\n", + "x_{n-1,0}& x_{n-1,1} &x_{n-1,2}& \\dots & \\dots &x_{n-1,n-1}\\\\\n", + "\\end{bmatrix}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and without loss of generality we rewrite again our equations as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y} = \\boldsymbol{X}\\boldsymbol{\\beta}+\\boldsymbol{\\epsilon}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The left-hand side of this equation is kwown. Our error vector $\\boldsymbol{\\epsilon}$ and the parameter vector $\\boldsymbol{\\beta}$ are our unknow quantities. How can we obtain the optimal set of $\\beta_i$ values? \n", + "\n", + "We have defined the matrix $\\boldsymbol{X}$ via the equations" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "y_0&=\\beta_0x_{00}+\\beta_1x_{01}+\\beta_2x_{02}+\\dots+\\beta_{n-1}x_{0n-1}+\\epsilon_0\\\\\n", + "y_1&=\\beta_0x_{10}+\\beta_1x_{11}+\\beta_2x_{12}+\\dots+\\beta_{n-1}x_{1n-1}+\\epsilon_1\\\\\n", + "y_2&=\\beta_0x_{20}+\\beta_1x_{21}+\\beta_2x_{22}+\\dots+\\beta_{n-1}x_{2n-1}+\\epsilon_1\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{i}&=\\beta_0x_{i0}+\\beta_1x_{i1}+\\beta_2x_{i2}+\\dots+\\beta_{n-1}x_{in-1}+\\epsilon_1\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{n-1}&=\\beta_0x_{n-1,0}+\\beta_1x_{n-1,2}+\\beta_2x_{n-1,2}+\\dots+\\beta_{n-1}x_{n-1,n-1}+\\epsilon_{n-1}.\\\\\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "As we noted above, we stayed with a system with the design matrix \n", + " $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times n}$, that is we have $p=n$. For reasons to come later (algorithmic arguments) we will hereafter define \n", + "our matrix as $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$, with the predictors refering to the column numbers and the entries $n$ being the row elements.\n", + "\n", + "In our [introductory notes](https://compphysics.github.io/MachineLearning/doc/pub/How2ReadData/html/How2ReadData.html) we looked at the so-called [liquid drop model](https://en.wikipedia.org/wiki/Semi-empirical_mass_formula). Let us remind ourselves about what we did by looking at the code.\n", + "\n", + "We restate the parts of the code we are most interested in." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from IPython.display import display\n", + "import os\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"MassEval2016.dat\"),'r')\n", + "\n", + "\n", + "# Read the experimental data with Pandas\n", + "Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11),\n", + " names=('N', 'Z', 'A', 'Element', 'Ebinding'),\n", + " widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),\n", + " header=39,\n", + " index_col=False)\n", + "\n", + "# Extrapolated values are indicated by '#' in place of the decimal place, so\n", + "# the Ebinding column won't be numeric. Coerce to float and drop these entries.\n", + "Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce')\n", + "Masses = Masses.dropna()\n", + "# Convert from keV to MeV.\n", + "Masses['Ebinding'] /= 1000\n", + "\n", + "# Group the DataFrame by nucleon number, A.\n", + "Masses = Masses.groupby('A')\n", + "# Find the rows of the grouped DataFrame with the maximum binding energy.\n", + "Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()])\n", + "A = Masses['A']\n", + "Z = Masses['Z']\n", + "N = Masses['N']\n", + "Element = Masses['Element']\n", + "Energies = Masses['Ebinding']\n", + "\n", + "# Now we set up the design matrix X\n", + "X = np.zeros((len(A),5))\n", + "X[:,0] = 1\n", + "X[:,1] = A\n", + "X[:,2] = A**(2.0/3.0)\n", + "X[:,3] = A**(-1.0/3.0)\n", + "X[:,4] = A**(-1.0)\n", + "# Then nice printout using pandas\n", + "DesignMatrix = pd.DataFrame(X)\n", + "DesignMatrix.index = A\n", + "DesignMatrix.columns = ['1', 'A', 'A^(2/3)', 'A^(-1/3)', '1/A']\n", + "display(DesignMatrix)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With $\\boldsymbol{\\beta}\\in {\\mathbb{R}}^{p\\times 1}$, it means that we will hereafter write our equations for the approximation as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\tilde{y}}= \\boldsymbol{X}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "throughout these lectures. \n", + "\n", + "With the above we use the design matrix to define the approximation $\\boldsymbol{\\tilde{y}}$ via the unknown quantity $\\boldsymbol{\\beta}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\tilde{y}}= \\boldsymbol{X}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and in order to find the optimal parameters $\\beta_i$ instead of solving the above linear algebra problem, we define a function which gives a measure of the spread between the values $y_i$ (which represent hopefully the exact values) and the parameterized values $\\tilde{y}_i$, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta})=\\frac{1}{n}\\sum_{i=0}^{n-1}\\left(y_i-\\tilde{y}_i\\right)^2=\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)\\right\\},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or using the matrix $\\boldsymbol{X}$ and in a more compact matrix-vector notation as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta})=\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)\\right\\}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This function is one possible way to define the so-called cost function.\n", + "\n", + "\n", + "\n", + "It is also common to define\n", + "the function $C$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta})=\\frac{1}{2n}\\sum_{i=0}^{n-1}\\left(y_i-\\tilde{y}_i\\right)^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "since when taking the first derivative with respect to the unknown parameters $\\beta$, the factor of $2$ cancels out. \n", + "\n", + "The function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta})=\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)\\right\\},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "can be linked to the variance of the quantity $y_i$ if we interpret the latter as the mean value. \n", + "When linking (see the discussion below) with the maximum likelihood approach below, we will indeed interpret $y_i$ as a mean value" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y_{i}=\\langle y_i \\rangle = \\beta_0x_{i,0}+\\beta_1x_{i,1}+\\beta_2x_{i,2}+\\dots+\\beta_{n-1}x_{i,n-1}+\\epsilon_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\langle y_i \\rangle$ is the mean value. Keep in mind also that\n", + "till now we have treated $y_i$ as the exact value. Normally, the\n", + "response (dependent or outcome) variable $y_i$ the outcome of a\n", + "numerical experiment or another type of experiment and is thus only an\n", + "approximation to the true value. It is then always accompanied by an\n", + "error estimate, often limited to a statistical error estimate given by\n", + "the standard deviation discussed earlier. In the discussion here we\n", + "will treat $y_i$ as our exact value for the response variable.\n", + "\n", + "In order to find the parameters $\\beta_i$ we will then minimize the spread of $C(\\boldsymbol{\\beta})$, that is we are going to solve the problem" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in\n", + "{\\mathbb{R}}^{p}}}\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)\\right\\}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In practical terms it means we will require" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial C(\\boldsymbol{\\beta})}{\\partial \\beta_j} = \\frac{\\partial }{\\partial \\beta_j}\\left[ \\frac{1}{n}\\sum_{i=0}^{n-1}\\left(y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}\\right)^2\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which results in" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial C(\\boldsymbol{\\beta})}{\\partial \\beta_j} = -\\frac{2}{n}\\left[ \\sum_{i=0}^{n-1}x_{ij}\\left(y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}\\right)\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or in a matrix-vector form as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial C(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can rewrite" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial C(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{y} = \\boldsymbol{X}^T\\boldsymbol{X}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and if the matrix $\\boldsymbol{X}^T\\boldsymbol{X}$ is invertible we have the solution" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta} =\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We note also that since our design matrix is defined as $\\boldsymbol{X}\\in\n", + "{\\mathbb{R}}^{n\\times p}$, the product $\\boldsymbol{X}^T\\boldsymbol{X} \\in\n", + "{\\mathbb{R}}^{p\\times p}$. In the above case we have that $p \\ll n$,\n", + "in our case $p=5$ meaning that we end up with inverting a small\n", + "$5\\times 5$ matrix. This is a rather common situation, in many cases we end up with low-dimensional\n", + "matrices to invert. The methods discussed here and for many other\n", + "supervised learning algorithms like classification with logistic\n", + "regression or support vector machines, exhibit dimensionalities which\n", + "allow for the usage of direct linear algebra methods such as **LU** decomposition or **Singular Value Decomposition** (SVD) for finding the inverse of the matrix\n", + "$\\boldsymbol{X}^T\\boldsymbol{X}$. \n", + "\n", + "**Small question**: Do you think the example we have at hand here (the nuclear binding energies) can lead to problems in inverting the matrix $\\boldsymbol{X}^T\\boldsymbol{X}$? What kind of problems can we expect? \n", + "\n", + "\n", + "The following matrix and vector relation will be useful here and for the rest of the course. Vectors are always written as boldfaced lower case letters and \n", + "matrices as upper case boldfaced letters." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "4\n", + "8\n", + " \n", + "<\n", + "<\n", + "<\n", + "!\n", + "!\n", + "M\n", + "A\n", + "T\n", + "H\n", + "_\n", + "B\n", + "L\n", + "O\n", + "C\n", + "K" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "4\n", + "9\n", + " \n", + "<\n", + "<\n", + "<\n", + "!\n", + "!\n", + "M\n", + "A\n", + "T\n", + "H\n", + "_\n", + "B\n", + "L\n", + "O\n", + "C\n", + "K" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "5\n", + "0\n", + " \n", + "<\n", + "<\n", + "<\n", + "!\n", + "!\n", + "M\n", + "A\n", + "T\n", + "H\n", + "_\n", + "B\n", + "L\n", + "O\n", + "C\n", + "K" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\log{\\vert\\boldsymbol{A}\\vert}}{\\partial \\boldsymbol{A}} = (\\boldsymbol{A}^{-1})^T.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The residuals $\\boldsymbol{\\epsilon}$ are in turn given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\epsilon} = \\boldsymbol{y}-\\boldsymbol{\\tilde{y}} = \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and with" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)= 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{\\epsilon}=\\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)= 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "meaning that the solution for $\\boldsymbol{\\beta}$ is the one which minimizes the residuals. Later we will link this with the maximum likelihood approach.\n", + "\n", + "\n", + "Let us now return to our nuclear binding energies and simply code the above equations. \n", + "\n", + "\n", + "It is rather straightforward to implement the matrix inversion and obtain the parameters $\\boldsymbol{\\beta}$. After having defined the matrix $\\boldsymbol{X}$ we simply need to \n", + "write" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# matrix inversion to find beta\n", + "beta = np.linalg.inv(X.T.dot(X)).dot(X.T).dot(Energies)\n", + "# and then make the prediction\n", + "ytilde = X @ beta" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Alternatively, you can use the least squares functionality in **Numpy** as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "fit = np.linalg.lstsq(X, Energies, rcond =None)[0]\n", + "ytildenp = np.dot(fit,X.T)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "And finally we plot our fit with and compare with data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "Masses['Eapprox'] = ytilde\n", + "# Generate a plot comparing the experimental with the fitted values values.\n", + "fig, ax = plt.subplots()\n", + "ax.set_xlabel(r'$A = N + Z$')\n", + "ax.set_ylabel(r'$E_\\mathrm{bind}\\,/\\mathrm{MeV}$')\n", + "ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,\n", + " label='Ame2016')\n", + "ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',\n", + " label='Fit')\n", + "ax.legend()\n", + "save_fig(\"Masses2016OLS\")\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can easily test our fit by computing the $R2$ score that we discussed in connection with the functionality of **Scikit-Learn** in the introductory slides.\n", + "Since we are not using **Scikit-Learn** here we can define our own $R2$ function as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "def R2(y_data, y_model):\n", + " return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and we would be using it as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "print(R2(Energies,ytilde))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can easily add our **MSE** score as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "\n", + "print(MSE(Energies,ytilde))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and finally the relative error as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "def RelativeError(y_data,y_model):\n", + " return abs((y_data-y_model)/y_data)\n", + "print(RelativeError(Energies, ytilde))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### The $\\chi^2$ function\n", + "\n", + "Normally, the response (dependent or outcome) variable $y_i$ is the\n", + "outcome of a numerical experiment or another type of experiment and is\n", + "thus only an approximation to the true value. It is then always\n", + "accompanied by an error estimate, often limited to a statistical error\n", + "estimate given by the standard deviation discussed earlier. In the\n", + "discussion here we will treat $y_i$ as our exact value for the\n", + "response variable.\n", + "\n", + "Introducing the standard deviation $\\sigma_i$ for each measurement\n", + "$y_i$, we define now the $\\chi^2$ function (omitting the $1/n$ term)\n", + "as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\chi^2(\\boldsymbol{\\beta})=\\frac{1}{n}\\sum_{i=0}^{n-1}\\frac{\\left(y_i-\\tilde{y}_i\\right)^2}{\\sigma_i^2}=\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)^T\\frac{1}{\\boldsymbol{\\Sigma^2}}\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)\\right\\},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where the matrix $\\boldsymbol{\\Sigma}$ is a diagonal matrix with $\\sigma_i$ as matrix elements. \n", + "\n", + "\n", + "In order to find the parameters $\\beta_i$ we will then minimize the spread of $\\chi^2(\\boldsymbol{\\beta})$ by requiring" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_j} = \\frac{\\partial }{\\partial \\beta_j}\\left[ \\frac{1}{n}\\sum_{i=0}^{n-1}\\left(\\frac{y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}}{\\sigma_i}\\right)^2\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which results in" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_j} = -\\frac{2}{n}\\left[ \\sum_{i=0}^{n-1}\\frac{x_{ij}}{\\sigma_i}\\left(\\frac{y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}}{\\sigma_i}\\right)\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or in a matrix-vector form as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{A}^T\\left( \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{\\beta}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have defined the matrix $\\boldsymbol{A} =\\boldsymbol{X}/\\boldsymbol{\\Sigma}$ with matrix elements $a_{ij} = x_{ij}/\\sigma_i$ and the vector $\\boldsymbol{b}$ with elements $b_i = y_i/\\sigma_i$. \n", + "\n", + "We can rewrite" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{A}^T\\left( \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{\\beta}\\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}^T\\boldsymbol{b} = \\boldsymbol{A}^T\\boldsymbol{A}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and if the matrix $\\boldsymbol{A}^T\\boldsymbol{A}$ is invertible we have the solution" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta} =\\left(\\boldsymbol{A}^T\\boldsymbol{A}\\right)^{-1}\\boldsymbol{A}^T\\boldsymbol{b}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we then introduce the matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{H} = \\left(\\boldsymbol{A}^T\\boldsymbol{A}\\right)^{-1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we have then the following expression for the parameters $\\beta_j$ (the matrix elements of $\\boldsymbol{H}$ are $h_{ij}$)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_j = \\sum_{k=0}^{p-1}h_{jk}\\sum_{i=0}^{n-1}\\frac{y_i}{\\sigma_i}\\frac{x_{ik}}{\\sigma_i} = \\sum_{k=0}^{p-1}h_{jk}\\sum_{i=0}^{n-1}b_ia_{ik}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We state without proof the expression for the uncertainty in the parameters $\\beta_j$ as (we leave this as an exercise)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\sigma^2(\\beta_j) = \\sum_{i=0}^{n-1}\\sigma_i^2\\left( \\frac{\\partial \\beta_j}{\\partial y_i}\\right)^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "resulting in" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\sigma^2(\\beta_j) = \\left(\\sum_{k=0}^{p-1}h_{jk}\\sum_{i=0}^{n-1}a_{ik}\\right)\\left(\\sum_{l=0}^{p-1}h_{jl}\\sum_{m=0}^{n-1}a_{ml}\\right) = h_{jj}!\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The first step here is to approximate the function $y$ with a first-order polynomial, that is we write" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y=y(x) \\rightarrow y(x_i) \\approx \\beta_0+\\beta_1 x_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "By computing the derivatives of $\\chi^2$ with respect to $\\beta_0$ and $\\beta_1$ show that these are given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_0} = -2\\left[ \\frac{1}{n}\\sum_{i=0}^{n-1}\\left(\\frac{y_i-\\beta_0-\\beta_1x_{i}}{\\sigma_i^2}\\right)\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_1} = -\\frac{2}{n}\\left[ \\sum_{i=0}^{n-1}x_i\\left(\\frac{y_i-\\beta_0-\\beta_1x_{i}}{\\sigma_i^2}\\right)\\right]=0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "For a linear fit (a first-order polynomial) we don't need to invert a matrix!! \n", + "Defining" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma = \\sum_{i=0}^{n-1}\\frac{1}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma_x = \\sum_{i=0}^{n-1}\\frac{x_{i}}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma_y = \\sum_{i=0}^{n-1}\\left(\\frac{y_i}{\\sigma_i^2}\\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma_{xx} = \\sum_{i=0}^{n-1}\\frac{x_ix_{i}}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma_{xy} = \\sum_{i=0}^{n-1}\\frac{y_ix_{i}}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_0 = \\frac{\\gamma_{xx}\\gamma_y-\\gamma_x\\gamma_y}{\\gamma\\gamma_{xx}-\\gamma_x^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_1 = \\frac{\\gamma_{xy}\\gamma-\\gamma_x\\gamma_y}{\\gamma\\gamma_{xx}-\\gamma_x^2}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This approach (different linear and non-linear regression) suffers\n", + "often from both being underdetermined and overdetermined in the\n", + "unknown coefficients $\\beta_i$. A better approach is to use the\n", + "Singular Value Decomposition (SVD) method discussed below. Or using\n", + "Lasso and Ridge regression. See below.\n", + "\n", + "\n", + "### Fitting an Equation of State for Dense Nuclear Matter\n", + "\n", + "Before we continue, let us introduce yet another example. We are going to fit the\n", + "nuclear equation of state using results from many-body calculations.\n", + "The equation of state we have made available here, as function of\n", + "density, has been derived using modern nucleon-nucleon potentials with\n", + "[the addition of three-body\n", + "forces](https://www.sciencedirect.com/science/article/pii/S0370157399001106). This\n", + "time the file is presented as a standard **csv** file.\n", + "\n", + "The beginning of the Python code here is similar to what you have seen\n", + "before, with the same initializations and declarations. We use also\n", + "**pandas** again, rather extensively in order to organize our data.\n", + "\n", + "The difference now is that we use **Scikit-Learn's** regression tools\n", + "instead of our own matrix inversion implementation. Furthermore, we\n", + "sneak in **Ridge** regression (to be discussed below) which includes a\n", + "hyperparameter $\\lambda$, also to be explained below." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "import matplotlib.pyplot as plt\n", + "import sklearn.linear_model as skl\n", + "from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"EoS.csv\"),'r')\n", + "\n", + "# Read the EoS data as csv file and organize the data into two arrays with density and energies\n", + "EoS = pd.read_csv(infile, names=('Density', 'Energy'))\n", + "EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')\n", + "EoS = EoS.dropna()\n", + "Energies = EoS['Energy']\n", + "Density = EoS['Density']\n", + "# The design matrix now as function of various polytrops\n", + "X = np.zeros((len(Density),4))\n", + "X[:,3] = Density**(4.0/3.0)\n", + "X[:,2] = Density\n", + "X[:,1] = Density**(2.0/3.0)\n", + "X[:,0] = 1\n", + "\n", + "# We use now Scikit-Learn's linear regressor and ridge regressor\n", + "# OLS part\n", + "clf = skl.LinearRegression().fit(X, Energies)\n", + "ytilde = clf.predict(X)\n", + "EoS['Eols'] = ytilde\n", + "# The mean squared error \n", + "print(\"Mean squared error: %.2f\" % mean_squared_error(Energies, ytilde))\n", + "# Explained variance score: 1 is perfect prediction \n", + "print('Variance score: %.2f' % r2_score(Energies, ytilde))\n", + "# Mean absolute error \n", + "print('Mean absolute error: %.2f' % mean_absolute_error(Energies, ytilde))\n", + "print(clf.coef_, clf.intercept_)\n", + "\n", + "# The Ridge regression with a hyperparameter lambda = 0.1\n", + "_lambda = 0.1\n", + "clf_ridge = skl.Ridge(alpha=_lambda).fit(X, Energies)\n", + "yridge = clf_ridge.predict(X)\n", + "EoS['Eridge'] = yridge\n", + "# The mean squared error \n", + "print(\"Mean squared error: %.2f\" % mean_squared_error(Energies, yridge))\n", + "# Explained variance score: 1 is perfect prediction \n", + "print('Variance score: %.2f' % r2_score(Energies, yridge))\n", + "# Mean absolute error \n", + "print('Mean absolute error: %.2f' % mean_absolute_error(Energies, yridge))\n", + "print(clf_ridge.coef_, clf_ridge.intercept_)\n", + "\n", + "fig, ax = plt.subplots()\n", + "ax.set_xlabel(r'$\\rho[\\mathrm{fm}^{-3}]$')\n", + "ax.set_ylabel(r'Energy per particle')\n", + "ax.plot(EoS['Density'], EoS['Energy'], alpha=0.7, lw=2,\n", + " label='Theoretical data')\n", + "ax.plot(EoS['Density'], EoS['Eols'], alpha=0.7, lw=2, c='m',\n", + " label='OLS')\n", + "ax.plot(EoS['Density'], EoS['Eridge'], alpha=0.7, lw=2, c='g',\n", + " label='Ridge $\\lambda = 0.1$')\n", + "ax.legend()\n", + "save_fig(\"EoSfitting\")\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The above simple polynomial in density $\\rho$ gives an excellent fit\n", + "to the data. \n", + "\n", + "We note also that there is a small deviation between the\n", + "standard OLS and the Ridge regression at higher densities. We discuss this in more detail\n", + "below.\n", + "\n", + "\n", + "## Splitting our Data in Training and Test data\n", + "\n", + "It is normal in essentially all Machine Learning studies to split the\n", + "data in a training set and a test set (sometimes also an additional\n", + "validation set). **Scikit-Learn** has an own function for this. There\n", + "is no explicit recipe for how much data should be included as training\n", + "data and say test data. An accepted rule of thumb is to use\n", + "approximately $2/3$ to $4/5$ of the data as training data. We will\n", + "postpone a discussion of this splitting to the end of these notes and\n", + "our discussion of the so-called **bias-variance** tradeoff. Here we\n", + "limit ourselves to repeat the above equation of state fitting example\n", + "but now splitting the data into a training set and a test set." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import train_test_split\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "def R2(y_data, y_model):\n", + " return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)\n", + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "\n", + "infile = open(data_path(\"EoS.csv\"),'r')\n", + "\n", + "# Read the EoS data as csv file and organized into two arrays with density and energies\n", + "EoS = pd.read_csv(infile, names=('Density', 'Energy'))\n", + "EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')\n", + "EoS = EoS.dropna()\n", + "Energies = EoS['Energy']\n", + "Density = EoS['Density']\n", + "# The design matrix now as function of various polytrops\n", + "X = np.zeros((len(Density),5))\n", + "X[:,0] = 1\n", + "X[:,1] = Density**(2.0/3.0)\n", + "X[:,2] = Density\n", + "X[:,3] = Density**(4.0/3.0)\n", + "X[:,4] = Density**(5.0/3.0)\n", + "# We split the data in test and training data\n", + "X_train, X_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2)\n", + "# matrix inversion to find beta\n", + "beta = np.linalg.inv(X_train.T.dot(X_train)).dot(X_train.T).dot(y_train)\n", + "# and then make the prediction\n", + "ytilde = X_train @ beta\n", + "print(\"Training R2\")\n", + "print(R2(y_train,ytilde))\n", + "print(\"Training MSE\")\n", + "print(MSE(y_train,ytilde))\n", + "ypredict = X_test @ beta\n", + "print(\"Test R2\")\n", + "print(R2(y_test,ypredict))\n", + "print(\"Test MSE\")\n", + "print(MSE(y_test,ypredict))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## The Boston housing data example\n", + "\n", + "The Boston housing \n", + "data set was originally a part of UCI Machine Learning Repository\n", + "and has been removed now. The data set is now included in **Scikit-Learn**'s \n", + "library. There are 506 samples and 13 feature (predictor) variables\n", + "in this data set. The objective is to predict the value of prices of\n", + "the house using the features (predictors) listed here.\n", + "\n", + "The features/predictors are\n", + "1. CRIM: Per capita crime rate by town\n", + "\n", + "2. ZN: Proportion of residential land zoned for lots over 25000 square feet\n", + "\n", + "3. INDUS: Proportion of non-retail business acres per town\n", + "\n", + "4. CHAS: Charles River dummy variable (= 1 if tract bounds river; 0 otherwise)\n", + "\n", + "5. NOX: Nitric oxide concentration (parts per 10 million)\n", + "\n", + "6. RM: Average number of rooms per dwelling\n", + "\n", + "7. AGE: Proportion of owner-occupied units built prior to 1940\n", + "\n", + "8. DIS: Weighted distances to five Boston employment centers\n", + "\n", + "9. RAD: Index of accessibility to radial highways\n", + "\n", + "10. TAX: Full-value property tax rate per USD10000\n", + "\n", + "11. B: $1000(Bk - 0.63)^2$, where $Bk$ is the proportion of [people of African American descent] by town\n", + "\n", + "12. LSTAT: Percentage of lower status of the population\n", + "\n", + "13. MEDV: Median value of owner-occupied homes in USD 1000s\n", + "\n", + "## Housing data, the code\n", + "We start by importing the libraries" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt \n", + "\n", + "import pandas as pd \n", + "import seaborn as sns" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and load the Boston Housing DataSet from **Scikit-Learn**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from sklearn.datasets import load_boston\n", + "\n", + "boston_dataset = load_boston()\n", + "\n", + "# boston_dataset is a dictionary\n", + "# let's check what it contains\n", + "boston_dataset.keys()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Then we invoke Pandas" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "boston = pd.DataFrame(boston_dataset.data, columns=boston_dataset.feature_names)\n", + "boston.head()\n", + "boston['MEDV'] = boston_dataset.target" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and preprocess the data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# check for missing values in all the columns\n", + "boston.isnull().sum()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can then visualize the data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# set the size of the figure\n", + "sns.set(rc={'figure.figsize':(11.7,8.27)})\n", + "\n", + "# plot a histogram showing the distribution of the target values\n", + "sns.distplot(boston['MEDV'], bins=30)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "It is now useful to look at the correlation matrix" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# compute the pair wise correlation for all columns \n", + "correlation_matrix = boston.corr().round(2)\n", + "# use the heatmap function from seaborn to plot the correlation matrix\n", + "# annot = True to print the values inside the square\n", + "sns.heatmap(data=correlation_matrix, annot=True)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "From the above coorelation plot we can see that **MEDV** is strongly correlated to **LSTAT** and **RM**. We see also that **RAD** and **TAX** are stronly correlated, but we don't include this in our features together to avoid multi-colinearity" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "plt.figure(figsize=(20, 5))\n", + "\n", + "features = ['LSTAT', 'RM']\n", + "target = boston['MEDV']\n", + "\n", + "for i, col in enumerate(features):\n", + " plt.subplot(1, len(features) , i+1)\n", + " x = boston[col]\n", + " y = target\n", + " plt.scatter(x, y, marker='o')\n", + " plt.title(col)\n", + " plt.xlabel(col)\n", + " plt.ylabel('MEDV')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Now we start training our model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "X = pd.DataFrame(np.c_[boston['LSTAT'], boston['RM']], columns = ['LSTAT','RM'])\n", + "Y = boston['MEDV']" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We split the data into training and test sets" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from sklearn.model_selection import train_test_split\n", + "\n", + "# splits the training and test data set in 80% : 20%\n", + "# assign random_state to any value.This ensures consistency.\n", + "X_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size = 0.2, random_state=5)\n", + "print(X_train.shape)\n", + "print(X_test.shape)\n", + "print(Y_train.shape)\n", + "print(Y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Then we use the linear regression functionality from **Scikit-Learn**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from sklearn.linear_model import LinearRegression\n", + "from sklearn.metrics import mean_squared_error, r2_score\n", + "\n", + "lin_model = LinearRegression()\n", + "lin_model.fit(X_train, Y_train)\n", + "\n", + "# model evaluation for training set\n", + "\n", + "y_train_predict = lin_model.predict(X_train)\n", + "rmse = (np.sqrt(mean_squared_error(Y_train, y_train_predict)))\n", + "r2 = r2_score(Y_train, y_train_predict)\n", + "\n", + "print(\"The model performance for training set\")\n", + "print(\"--------------------------------------\")\n", + "print('RMSE is {}'.format(rmse))\n", + "print('R2 score is {}'.format(r2))\n", + "print(\"\\n\")\n", + "\n", + "# model evaluation for testing set\n", + "\n", + "y_test_predict = lin_model.predict(X_test)\n", + "# root mean square error of the model\n", + "rmse = (np.sqrt(mean_squared_error(Y_test, y_test_predict)))\n", + "\n", + "# r-squared score of the model\n", + "r2 = r2_score(Y_test, y_test_predict)\n", + "\n", + "print(\"The model performance for testing set\")\n", + "print(\"--------------------------------------\")\n", + "print('RMSE is {}'.format(rmse))\n", + "print('R2 score is {}'.format(r2))" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# plotting the y_test vs y_pred\n", + "# ideally should have been a straight line\n", + "plt.scatter(Y_test, y_test_predict)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Reducing the number of degrees of freedom, overarching view\n", + "\n", + "Many Machine Learning problems involve thousands or even millions of\n", + "features for each training instance. Not only does this make training\n", + "extremely slow, it can also make it much harder to find a good\n", + "solution, as we will see. This problem is often referred to as the\n", + "curse of dimensionality. Fortunately, in real-world problems, it is\n", + "often possible to reduce the number of features considerably, turning\n", + "an intractable problem into a tractable one.\n", + "\n", + "Later we will discuss some of the most popular dimensionality reduction\n", + "techniques: the principal component analysis (PCA), Kernel PCA, and\n", + "Locally Linear Embedding (LLE). \n", + "\n", + "\n", + "Principal component analysis and its various variants deal with the\n", + "problem of fitting a low-dimensional [affine\n", + "subspace](https://en.wikipedia.org/wiki/Affine_space) to a set of of\n", + "data points in a high-dimensional space. With its family of methods it\n", + "is one of the most used tools in data modeling, compression and\n", + "visualization.\n", + "\n", + "\n", + "Before we proceed however, we will discuss how to preprocess our\n", + "data. Till now and in connection with our previous examples we have\n", + "not met so many cases where we are too sensitive to the scaling of our\n", + "data. Normally the data may need a rescaling and/or may be sensitive\n", + "to extreme values. Scaling the data renders our inputs much more\n", + "suitable for the algorithms we want to employ.\n", + "\n", + "**Scikit-Learn** has several functions which allow us to rescale the\n", + "data, normally resulting in much better results in terms of various\n", + "accuracy scores. The **StandardScaler** function in **Scikit-Learn**\n", + "ensures that for each feature/predictor we study the mean value is\n", + "zero and the variance is one (every column in the design/feature\n", + "matrix). This scaling has the drawback that it does not ensure that\n", + "we have a particular maximum or minimum in our data set. Another\n", + "function included in **Scikit-Learn** is the **MinMaxScaler** which\n", + "ensures that all features are exactly between $0$ and $1$. The\n", + "\n", + "\n", + "The **Normalizer** scales each data\n", + "point such that the feature vector has a euclidean length of one. In other words, it\n", + "projects a data point on the circle (or sphere in the case of higher dimensions) with a\n", + "radius of 1. This means every data point is scaled by a different number (by the\n", + "inverse of it’s length).\n", + "This normalization is often used when only the direction (or angle) of the data matters,\n", + "not the length of the feature vector.\n", + "\n", + "The **RobustScaler** works similarly to the StandardScaler in that it\n", + "ensures statistical properties for each feature that guarantee that\n", + "they are on the same scale. However, the RobustScaler uses the median\n", + "and quartiles, instead of mean and variance. This makes the\n", + "RobustScaler ignore data points that are very different from the rest\n", + "(like measurement errors). These odd data points are also called\n", + "outliers, and might often lead to trouble for other scaling\n", + "techniques.\n", + "\n", + "\n", + "### Simple preprocessing examples, Franke function and regression" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "import sklearn.linear_model as skl\n", + "from sklearn.metrics import mean_squared_error\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.preprocessing import MinMaxScaler, StandardScaler, Normalizer\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "\n", + "def FrankeFunction(x,y):\n", + "\tterm1 = 0.75*np.exp(-(0.25*(9*x-2)**2) - 0.25*((9*y-2)**2))\n", + "\tterm2 = 0.75*np.exp(-((9*x+1)**2)/49.0 - 0.1*(9*y+1))\n", + "\tterm3 = 0.5*np.exp(-(9*x-7)**2/4.0 - 0.25*((9*y-3)**2))\n", + "\tterm4 = -0.2*np.exp(-(9*x-4)**2 - (9*y-7)**2)\n", + "\treturn term1 + term2 + term3 + term4\n", + "\n", + "\n", + "def create_X(x, y, n ):\n", + "\tif len(x.shape) > 1:\n", + "\t\tx = np.ravel(x)\n", + "\t\ty = np.ravel(y)\n", + "\n", + "\tN = len(x)\n", + "\tl = int((n+1)*(n+2)/2)\t\t# Number of elements in beta\n", + "\tX = np.ones((N,l))\n", + "\n", + "\tfor i in range(1,n+1):\n", + "\t\tq = int((i)*(i+1)/2)\n", + "\t\tfor k in range(i+1):\n", + "\t\t\tX[:,q+k] = (x**(i-k))*(y**k)\n", + "\n", + "\treturn X\n", + "\n", + "\n", + "# Making meshgrid of datapoints and compute Franke's function\n", + "n = 5\n", + "N = 1000\n", + "x = np.sort(np.random.uniform(0, 1, N))\n", + "y = np.sort(np.random.uniform(0, 1, N))\n", + "z = FrankeFunction(x, y)\n", + "X = create_X(x, y, n=n) \n", + "# split in training and test data\n", + "X_train, X_test, y_train, y_test = train_test_split(X,z,test_size=0.2)\n", + "\n", + "\n", + "clf = skl.LinearRegression().fit(X_train, y_train)\n", + "\n", + "# The mean squared error and R2 score\n", + "print(\"MSE before scaling: {:.2f}\".format(mean_squared_error(clf.predict(X_test), y_test)))\n", + "print(\"R2 score before scaling {:.2f}\".format(clf.score(X_test,y_test)))\n", + "\n", + "scaler = StandardScaler()\n", + "scaler.fit(X_train)\n", + "X_train_scaled = scaler.transform(X_train)\n", + "X_test_scaled = scaler.transform(X_test)\n", + "\n", + "print(\"Feature min values before scaling:\\n {}\".format(X_train.min(axis=0)))\n", + "print(\"Feature max values before scaling:\\n {}\".format(X_train.max(axis=0)))\n", + "\n", + "print(\"Feature min values after scaling:\\n {}\".format(X_train_scaled.min(axis=0)))\n", + "print(\"Feature max values after scaling:\\n {}\".format(X_train_scaled.max(axis=0)))\n", + "\n", + "clf = skl.LinearRegression().fit(X_train_scaled, y_train)\n", + "\n", + "\n", + "print(\"MSE after scaling: {:.2f}\".format(mean_squared_error(clf.predict(X_test_scaled), y_test)))\n", + "print(\"R2 score for scaled data: {:.2f}\".format(clf.score(X_test_scaled,y_test)))" + ] + } + ], + "metadata": {}, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/doc/LectureNotes/_build/html/_sources/chapter2.ipynb b/doc/LectureNotes/_build/html/_sources/chapter2.ipynb new file mode 100644 index 000000000..721b43cde --- /dev/null +++ b/doc/LectureNotes/_build/html/_sources/chapter2.ipynb @@ -0,0 +1,1355 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Resampling Methods\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSept3.mp4?vrtx=view-as-webpage)\n", + "\n", + "\n", + "## Introduction\n", + "\n", + "Resampling methods are an indispensable tool in modern\n", + "statistics. They involve repeatedly drawing samples from a training\n", + "set and refitting a model of interest on each sample in order to\n", + "obtain additional information about the fitted model. For example, in\n", + "order to estimate the variability of a linear regression fit, we can\n", + "repeatedly draw different samples from the training data, fit a linear\n", + "regression to each new sample, and then examine the extent to which\n", + "the resulting fits differ. Such an approach may allow us to obtain\n", + "information that would not be available from fitting the model only\n", + "once using the original training sample.\n", + "\n", + "Two resampling methods are often used in Machine Learning analyses,\n", + "1. The **bootstrap method**\n", + "\n", + "2. and **Cross-Validation**\n", + "\n", + "In addition there are several other methods such as the Jackknife and the Blocking methods. We will discuss in particular\n", + "cross-validation and the bootstrap method. \n", + "\n", + "\n", + "Resampling approaches can be computationally expensive, because they\n", + "involve fitting the same statistical method multiple times using\n", + "different subsets of the training data. However, due to recent\n", + "advances in computing power, the computational requirements of\n", + "resampling methods generally are not prohibitive. In this chapter, we\n", + "discuss two of the most commonly used resampling methods,\n", + "cross-validation and the bootstrap. Both methods are important tools\n", + "in the practical application of many statistical learning\n", + "procedures. For example, cross-validation can be used to estimate the\n", + "test error associated with a given statistical learning method in\n", + "order to evaluate its performance, or to select the appropriate level\n", + "of flexibility. The process of evaluating a model’s performance is\n", + "known as model assessment, whereas the process of selecting the proper\n", + "level of flexibility for a model is known as model selection. The\n", + "bootstrap is widely used.\n", + "\n", + "\n", + "* Our simulations can be treated as *computer experiments*. This is particularly the case for Monte Carlo methods\n", + "\n", + "* The results can be analysed with the same statistical tools as we would use analysing experimental data.\n", + "\n", + "* As in all experiments, we are looking for expectation values and an estimate of how accurate they are, i.e., possible sources for errors.\n", + "\n", + "## Reminder on Statistics\n", + "\n", + "\n", + "* As in other experiments, many numerical experiments have two classes of errors:\n", + "\n", + " * Statistical errors\n", + "\n", + " * Systematical errors\n", + "\n", + "\n", + "* Statistical errors can be estimated using standard tools from statistics\n", + "\n", + "* Systematical errors are method specific and must be treated differently from case to case. \n", + "\n", + "The\n", + "advantage of doing linear regression is that we actually end up with\n", + "analytical expressions for several statistical quantities. \n", + "Standard least squares and Ridge regression allow us to\n", + "derive quantities like the variance and other expectation values in a\n", + "rather straightforward way.\n", + "\n", + "\n", + "It is assumed that $\\varepsilon_i\n", + "\\sim \\mathcal{N}(0, \\sigma^2)$ and the $\\varepsilon_{i}$ are\n", + "independent, i.e.:" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*} \n", + "\\mbox{Cov}(\\varepsilon_{i_1},\n", + "\\varepsilon_{i_2}) & = \\left\\{ \\begin{array}{lcc} \\sigma^2 & \\mbox{if}\n", + "& i_1 = i_2, \\\\ 0 & \\mbox{if} & i_1 \\not= i_2. \\end{array} \\right.\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The randomness of $\\varepsilon_i$ implies that\n", + "$\\mathbf{y}_i$ is also a random variable. In particular,\n", + "$\\mathbf{y}_i$ is normally distributed, because $\\varepsilon_i \\sim\n", + "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\beta}$ is a\n", + "non-random scalar. To specify the parameters of the distribution of\n", + "$\\mathbf{y}_i$ we need to calculate its first two moments. \n", + "\n", + "Recall that $\\boldsymbol{X}$ is a matrix of dimensionality $n\\times p$. The\n", + "notation above $\\mathbf{X}_{i,\\ast}$ means that we are looking at the\n", + "row number $i$ and perform a sum over all values $p$.\n", + "\n", + "\n", + "The assumption we have made here can be summarized as (and this is going to be useful when we discuss the bias-variance trade off)\n", + "that there exists a function $f(\\boldsymbol{x})$ and a normal distributed error $\\boldsymbol{\\varepsilon}\\sim \\mathcal{N}(0, \\sigma^2)$\n", + "which describe our data" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y} = f(\\boldsymbol{x})+\\boldsymbol{\\varepsilon}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We approximate this function with our model from the solution of the linear regression equations, that is our\n", + "function $f$ is approximated by $\\boldsymbol{\\tilde{y}}$ where we want to minimize $(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2$, our MSE, with" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\beta}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can calculate the expectation value of $\\boldsymbol{y}$ for a given element $i$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*} \n", + "\\mathbb{E}(y_i) & =\n", + "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}) + \\mathbb{E}(\\varepsilon_i)\n", + "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\beta, \n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "while\n", + "its variance is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*} \\mbox{Var}(y_i) & = \\mathbb{E} \\{ [y_i\n", + "- \\mathbb{E}(y_i)]^2 \\} \\, \\, \\, = \\, \\, \\, \\mathbb{E} ( y_i^2 ) -\n", + "[\\mathbb{E}(y_i)]^2 \\\\ & = \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\,\n", + "\\beta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \\\\ &\n", + "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2 \\varepsilon_i\n", + "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", + "\\ast} \\, \\beta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2\n", + "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} +\n", + "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \n", + "\\\\ & = \\mathbb{E}(\\varepsilon_i^2 ) \\, \\, \\, = \\, \\, \\,\n", + "\\mbox{Var}(\\varepsilon_i) \\, \\, \\, = \\, \\, \\, \\sigma^2. \n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", + "mean value $\\boldsymbol{X}\\boldsymbol{\\beta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD). \n", + "\n", + "\n", + "With the OLS expressions for the parameters $\\boldsymbol{\\beta}$ we can evaluate the expectation value" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}(\\boldsymbol{\\beta}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\beta}=\\boldsymbol{\\beta}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This means that the estimator of the regression parameters is unbiased.\n", + "\n", + "We can also calculate the variance\n", + "\n", + "The variance of $\\boldsymbol{\\beta}$ is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{eqnarray*}\n", + "\\mbox{Var}(\\boldsymbol{\\beta}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})] [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})]^{T} \\}\n", + "\\\\\n", + "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}]^{T} \\}\n", + "\\\\\n", + "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% \\\\\n", + "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% \\\\\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "\\\\\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% \\\\\n", + "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", + "% \\\\\n", + "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\boldsymbol{\\beta}^T\n", + "\\\\\n", + "& = & \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "\\, \\, \\, = \\, \\, \\, \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1},\n", + "\\end{eqnarray*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have used that $\\mathbb{E} (\\mathbf{Y} \\mathbf{Y}^{T}) =\n", + "\\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} +\n", + "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\beta}) = \\sigma^2\n", + "\\, (\\mathbf{X}^{T} \\mathbf{X})^{-1}$, one obtains an estimate of the\n", + "variance of the estimate of the $j$-th regression coefficient:\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 \\sqrt{\n", + "[(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} }$. This may be used to\n", + "construct a confidence interval for the estimates.\n", + "\n", + "\n", + "In a similar way, we can obtain analytical expressions for say the\n", + "expectation values of the parameters $\\boldsymbol{\\beta}$ and their variance\n", + "when we employ Ridge regression, allowing us again to define a confidence interval. \n", + "\n", + "It is rather straightforward to show that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\beta}^{\\mathrm{OLS}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We see clearly that \n", + "$\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\beta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", + "\n", + "We can also compute the variance as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\beta}$ goes to zero. \n", + "\n", + "With this, we can compute the difference" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\beta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The difference is non-negative definite since each component of the\n", + "matrix product is non-negative definite. \n", + "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\beta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below. \n", + "\n", + "\n", + "\n", + "## Resampling methods\n", + "\n", + "With all these analytical equations for both the OLS and Ridge\n", + "regression, we will now outline how to assess a given model. This will\n", + "lead us to a discussion of the so-called bias-variance tradeoff (see\n", + "below) and so-called resampling methods.\n", + "\n", + "One of the quantities we have discussed as a way to measure errors is\n", + "the mean-squared error (MSE), mainly used for fitting of continuous\n", + "functions. Another choice is the absolute error.\n", + "\n", + "In the discussions below we will focus on the MSE and in particular since we will split the data into test and training data,\n", + "we discuss the\n", + "1. prediction error or simply the **test error** $\\mathrm{Err_{Test}}$, where we have a fixed training set and the test error is the MSE arising from the data reserved for testing. We discuss also the \n", + "\n", + "2. training error $\\mathrm{Err_{Train}}$, which is the average loss over the training data.\n", + "\n", + "As our model becomes more and more complex, more of the training data tends to used. The training may thence adapt to more complicated structures in the data. This may lead to a decrease in the bias (see below for code example) and a slight increase of the variance for the test error.\n", + "For a certain level of complexity the test error will reach minimum, before starting to increase again. The\n", + "training error reaches a saturation.\n", + "\n", + "\n", + "\n", + "Two famous\n", + "resampling methods are the **independent bootstrap** and **the jackknife**. \n", + "\n", + "The jackknife is a special case of the independent bootstrap. Still, the jackknife was made\n", + "popular prior to the independent bootstrap. And as the popularity of\n", + "the independent bootstrap soared, new variants, such as **the dependent bootstrap**.\n", + "\n", + "The Jackknife and independent bootstrap work for\n", + "independent, identically distributed random variables.\n", + "If these conditions are not\n", + "satisfied, the methods will fail. Yet, it should be said that if the data are\n", + "independent, identically distributed, and we only want to estimate the\n", + "variance of $\\overline{X}$ (which often is the case), then there is no\n", + "need for bootstrapping. \n", + "\n", + "\n", + "The Jackknife works by making many replicas of the estimator $\\widehat{\\theta}$. \n", + "The jackknife is a resampling method where we systematically leave out one observation from the vector of observed values $\\boldsymbol{x} = (x_1,x_2,\\cdots,X_n)$. \n", + "Let $\\boldsymbol{x}_i$ denote the vector" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_i = (x_1,x_2,\\cdots,x_{i-1},x_{i+1},\\cdots,x_n),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which equals the vector $\\boldsymbol{x}$ with the exception that observation\n", + "number $i$ is left out. Using this notation, define\n", + "$\\widehat{\\theta}_i$ to be the estimator\n", + "$\\widehat{\\theta}$ computed using $\\vec{X}_i$." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from numpy import *\n", + "from numpy.random import randint, randn\n", + "from time import time\n", + "\n", + "def jackknife(data, stat):\n", + " n = len(data);t = zeros(n); inds = arange(n); t0 = time()\n", + " ## 'jackknifing' by leaving out an observation for each i \n", + " for i in range(n):\n", + " t[i] = stat(delete(data,i) )\n", + "\n", + " # analysis \n", + " print(\"Runtime: %g sec\" % (time()-t0)); print(\"Jackknife Statistics :\")\n", + " print(\"original bias std. error\")\n", + " print(\"%8g %14g %15g\" % (stat(data),(n-1)*mean(t)/n, (n*var(t))**.5))\n", + "\n", + " return t\n", + "\n", + "\n", + "# Returns mean of data samples \n", + "def stat(data):\n", + " return mean(data)\n", + "\n", + "\n", + "mu, sigma = 100, 15\n", + "datapoints = 10000\n", + "x = mu + sigma*random.randn(datapoints)\n", + "# jackknife returns the data sample \n", + "t = jackknife(x, stat)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### Bootstrap\n", + "\n", + "Bootstrapping is a nonparametric approach to statistical inference\n", + "that substitutes computation for more traditional distributional\n", + "assumptions and asymptotic results. Bootstrapping offers a number of\n", + "advantages: \n", + "1. The bootstrap is quite general, although there are some cases in which it fails. \n", + "\n", + "2. Because it does not require distributional assumptions (such as normally distributed errors), the bootstrap can provide more accurate inferences when the data are not well behaved or when the sample size is small. \n", + "\n", + "3. It is possible to apply the bootstrap to statistics with sampling distributions that are difficult to derive, even asymptotically. \n", + "\n", + "4. It is relatively simple to apply the bootstrap to complex data-collection plans (such as stratified and clustered samples).\n", + "\n", + "Since $\\widehat{\\theta} = \\widehat{\\theta}(\\boldsymbol{X})$ is a function of random variables,\n", + "$\\widehat{\\theta}$ itself must be a random variable. Thus it has\n", + "a pdf, call this function $p(\\boldsymbol{t})$. The aim of the bootstrap is to\n", + "estimate $p(\\boldsymbol{t})$ by the relative frequency of\n", + "$\\widehat{\\theta}$. You can think of this as using a histogram\n", + "in the place of $p(\\boldsymbol{t})$. If the relative frequency closely\n", + "resembles $p(\\vec{t})$, then using numerics, it is straight forward to\n", + "estimate all the interesting parameters of $p(\\boldsymbol{t})$ using point\n", + "estimators. \n", + "\n", + "\n", + "\n", + "In the case that $\\widehat{\\theta}$ has\n", + "more than one component, and the components are independent, we use the\n", + "same estimator on each component separately. If the probability\n", + "density function of $X_i$, $p(x)$, had been known, then it would have\n", + "been straight forward to do this by: \n", + "1. Drawing lots of numbers from $p(x)$, suppose we call one such set of numbers $(X_1^*, X_2^*, \\cdots, X_n^*)$. \n", + "\n", + "2. Then using these numbers, we could compute a replica of $\\widehat{\\theta}$ called $\\widehat{\\theta}^*$. \n", + "\n", + "By repeated use of (1) and (2), many\n", + "estimates of $\\widehat{\\theta}$ could have been obtained. The\n", + "idea is to use the relative frequency of $\\widehat{\\theta}^*$\n", + "(think of a histogram) as an estimate of $p(\\boldsymbol{t})$.\n", + "\n", + "\n", + "But\n", + "unless there is enough information available about the process that\n", + "generated $X_1,X_2,\\cdots,X_n$, $p(x)$ is in general\n", + "unknown. Therefore, [Efron in 1979](https://projecteuclid.org/euclid.aos/1176344552) asked the\n", + "question: What if we replace $p(x)$ by the relative frequency\n", + "of the observation $X_i$; if we draw observations in accordance with\n", + "the relative frequency of the observations, will we obtain the same\n", + "result in some asymptotic sense? The answer is yes.\n", + "\n", + "\n", + "Instead of generating the histogram for the relative\n", + "frequency of the observation $X_i$, just draw the values\n", + "$(X_1^*,X_2^*,\\cdots,X_n^*)$ with replacement from the vector\n", + "$\\boldsymbol{X}$. \n", + "\n", + "\n", + "The independent bootstrap works like this: \n", + "\n", + "1. Draw with replacement $n$ numbers for the observed variables $\\boldsymbol{x} = (x_1,x_2,\\cdots,x_n)$. \n", + "\n", + "2. Define a vector $\\boldsymbol{x}^*$ containing the values which were drawn from $\\boldsymbol{x}$. \n", + "\n", + "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\theta}^*$ by evaluating $\\widehat \\theta$ under the observations $\\boldsymbol{x}^*$. \n", + "\n", + "4. Repeat this process $k$ times. \n", + "\n", + "When you are done, you can draw a histogram of the relative frequency\n", + "of $\\widehat \\theta^*$. This is your estimate of the probability\n", + "distribution $p(t)$. Using this probability distribution you can\n", + "estimate any statistics thereof. In principle you never draw the\n", + "histogram of the relative frequency of $\\widehat{\\theta}^*$. Instead\n", + "you use the estimators corresponding to the statistic of interest. For\n", + "example, if you are interested in estimating the variance of $\\widehat\n", + "\\theta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", + "$\\widehat \\theta ^*$.\n", + "\n", + "\n", + "\n", + "The following code starts with a Gaussian distribution with mean value\n", + "$\\mu =100$ and variance $\\sigma=15$. We use this to generate the data\n", + "used in the bootstrap analysis. The bootstrap analysis returns a data\n", + "set after a given number of bootstrap operations (as many as we have\n", + "data points). This data set consists of estimated mean values for each\n", + "bootstrap operation. The histogram generated by the bootstrap method\n", + "shows that the distribution for these mean values is also a Gaussian,\n", + "centered around the mean value $\\mu=100$ but with standard deviation\n", + "$\\sigma/\\sqrt{n}$, where $n$ is the number of bootstrap samples (in\n", + "this case the same as the number of original data points). The value\n", + "of the standard deviation is what we expect from the central limit\n", + "theorem." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "%matplotlib inline\n", + "\n", + "from numpy import *\n", + "from numpy.random import randint, randn\n", + "from time import time\n", + "import matplotlib.mlab as mlab\n", + "import matplotlib.pyplot as plt\n", + "\n", + "# Returns mean of bootstrap samples \n", + "def stat(data):\n", + " return mean(data)\n", + "\n", + "# Bootstrap algorithm\n", + "def bootstrap(data, statistic, R):\n", + " t = zeros(R); n = len(data); inds = arange(n); t0 = time()\n", + " # non-parametric bootstrap \n", + " for i in range(R):\n", + " t[i] = statistic(data[randint(0,n,n)])\n", + "\n", + " # analysis \n", + " print(\"Runtime: %g sec\" % (time()-t0)); print(\"Bootstrap Statistics :\")\n", + " print(\"original bias std. error\")\n", + " print(\"%8g %8g %14g %15g\" % (statistic(data), std(data),mean(t),std(t)))\n", + " return t\n", + "\n", + "\n", + "mu, sigma = 100, 15\n", + "datapoints = 10000\n", + "x = mu + sigma*random.randn(datapoints)\n", + "# bootstrap returns the data sample \n", + "t = bootstrap(x, stat, datapoints)\n", + "# the histogram of the bootstrapped data \n", + "n, binsboot, patches = plt.hist(t, 50, normed=1, facecolor='red', alpha=0.75)\n", + "\n", + "# add a 'best fit' line \n", + "y = mlab.normpdf( binsboot, mean(t), std(t))\n", + "lt = plt.plot(binsboot, y, 'r--', linewidth=1)\n", + "plt.xlabel('Smarts')\n", + "plt.ylabel('Probability')\n", + "plt.axis([99.5, 100.6, 0, 3.0])\n", + "plt.grid(True)\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Various steps in cross-validation\n", + "\n", + "When the repetitive splitting of the data set is done randomly,\n", + "samples may accidently end up in a fast majority of the splits in\n", + "either training or test set. Such samples may have an unbalanced\n", + "influence on either model building or prediction evaluation. To avoid\n", + "this $k$-fold cross-validation structures the data splitting. The\n", + "samples are divided into $k$ more or less equally sized exhaustive and\n", + "mutually exclusive subsets. In turn (at each split) one of these\n", + "subsets plays the role of the test set while the union of the\n", + "remaining subsets constitutes the training set. Such a splitting\n", + "warrants a balanced representation of each sample in both training and\n", + "test set over the splits. Still the division into the $k$ subsets\n", + "involves a degree of randomness. This may be fully excluded when\n", + "choosing $k=n$. This particular case is referred to as leave-one-out\n", + "cross-validation (LOOCV). \n", + "\n", + "\n", + "* Define a range of interest for the penalty parameter.\n", + "\n", + "* Divide the data set into training and test set comprising samples $\\{1, \\ldots, n\\} \\setminus i$ and $\\{ i \\}$, respectively.\n", + "\n", + "* Fit the linear regression model by means of ridge estimation for each $\\lambda$ in the grid using the training set, and the corresponding estimate of the error variance $\\boldsymbol{\\sigma}_{-i}^2(\\lambda)$, as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "\\boldsymbol{\\beta}_{-i}(\\lambda) & = ( \\boldsymbol{X}_{-i, \\ast}^{T}\n", + "\\boldsymbol{X}_{-i, \\ast} + \\lambda \\boldsymbol{I}_{pp})^{-1}\n", + "\\boldsymbol{X}_{-i, \\ast}^{T} \\boldsymbol{y}_{-i}\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "* Evaluate the prediction performance of these models on the test set by $\\log\\{L[y_i, \\boldsymbol{X}_{i, \\ast}; \\boldsymbol{\\beta}_{-i}(\\lambda), \\boldsymbol{\\sigma}_{-i}^2(\\lambda)]\\}$. Or, by the prediction error $|y_i - \\boldsymbol{X}_{i, \\ast} \\boldsymbol{\\beta}_{-i}(\\lambda)|$, the relative error, the error squared or the R2 score function.\n", + "\n", + "* Repeat the first three steps such that each sample plays the role of the test set once.\n", + "\n", + "* Average the prediction performances of the test sets at each grid point of the penalty bias/parameter. It is an estimate of the prediction performance of the model corresponding to this value of the penalty parameter on novel data. It is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "\\frac{1}{n} \\sum_{i = 1}^n \\log\\{L[y_i, \\mathbf{X}_{i, \\ast}; \\boldsymbol{\\beta}_{-i}(\\lambda), \\boldsymbol{\\sigma}_{-i}^2(\\lambda)]\\}.\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "For the various values of $k$\n", + "\n", + "1. shuffle the dataset randomly.\n", + "\n", + "2. Split the dataset into $k$ groups.\n", + "\n", + "3. For each unique group:\n", + "\n", + "a. Decide which group to use as set for test data\n", + "\n", + "b. Take the remaining groups as a training data set\n", + "\n", + "c. Fit a model on the training set and evaluate it on the test set\n", + "\n", + "d. Retain the evaluation score and discard the model\n", + "\n", + "\n", + "5. Summarize the model using the sample of model evaluation scores\n", + "\n", + "The code here uses Ridge regression with cross-validation (CV) resampling and $k$-fold CV in order to fit a specific polynomial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import KFold\n", + "from sklearn.linear_model import Ridge\n", + "from sklearn.model_selection import cross_val_score\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "\n", + "# A seed just to ensure that the random numbers are the same for every run.\n", + "# Useful for eventual debugging.\n", + "np.random.seed(3155)\n", + "\n", + "# Generate the data.\n", + "nsamples = 100\n", + "x = np.random.randn(nsamples)\n", + "y = 3*x**2 + np.random.randn(nsamples)\n", + "\n", + "## Cross-validation on Ridge regression using KFold only\n", + "\n", + "# Decide degree on polynomial to fit\n", + "poly = PolynomialFeatures(degree = 6)\n", + "\n", + "# Decide which values of lambda to use\n", + "nlambdas = 500\n", + "lambdas = np.logspace(-3, 5, nlambdas)\n", + "\n", + "# Initialize a KFold instance\n", + "k = 5\n", + "kfold = KFold(n_splits = k)\n", + "\n", + "# Perform the cross-validation to estimate MSE\n", + "scores_KFold = np.zeros((nlambdas, k))\n", + "\n", + "i = 0\n", + "for lmb in lambdas:\n", + " ridge = Ridge(alpha = lmb)\n", + " j = 0\n", + " for train_inds, test_inds in kfold.split(x):\n", + " xtrain = x[train_inds]\n", + " ytrain = y[train_inds]\n", + "\n", + " xtest = x[test_inds]\n", + " ytest = y[test_inds]\n", + "\n", + " Xtrain = poly.fit_transform(xtrain[:, np.newaxis])\n", + " ridge.fit(Xtrain, ytrain[:, np.newaxis])\n", + "\n", + " Xtest = poly.fit_transform(xtest[:, np.newaxis])\n", + " ypred = ridge.predict(Xtest)\n", + "\n", + " scores_KFold[i,j] = np.sum((ypred - ytest[:, np.newaxis])**2)/np.size(ypred)\n", + "\n", + " j += 1\n", + " i += 1\n", + "\n", + "\n", + "estimated_mse_KFold = np.mean(scores_KFold, axis = 1)\n", + "\n", + "## Cross-validation using cross_val_score from sklearn along with KFold\n", + "\n", + "# kfold is an instance initialized above as:\n", + "# kfold = KFold(n_splits = k)\n", + "\n", + "estimated_mse_sklearn = np.zeros(nlambdas)\n", + "i = 0\n", + "for lmb in lambdas:\n", + " ridge = Ridge(alpha = lmb)\n", + "\n", + " X = poly.fit_transform(x[:, np.newaxis])\n", + " estimated_mse_folds = cross_val_score(ridge, X, y[:, np.newaxis], scoring='neg_mean_squared_error', cv=kfold)\n", + "\n", + " # cross_val_score return an array containing the estimated negative mse for every fold.\n", + " # we have to the the mean of every array in order to get an estimate of the mse of the model\n", + " estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)\n", + "\n", + " i += 1\n", + "\n", + "## Plot and compare the slightly different ways to perform cross-validation\n", + "\n", + "plt.figure()\n", + "\n", + "plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')\n", + "plt.plot(np.log10(lambdas), estimated_mse_KFold, 'r--', label = 'KFold')\n", + "\n", + "plt.xlabel('log10(lambda)')\n", + "plt.ylabel('mse')\n", + "\n", + "plt.legend()\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## The bias-variance tradeoff\n", + "\n", + "\n", + "We will discuss the bias-variance tradeoff in the context of\n", + "continuous predictions such as regression. However, many of the\n", + "intuitions and ideas discussed here also carry over to classification\n", + "tasks. Consider a dataset $\\mathcal{L}$ consisting of the data\n", + "$\\mathbf{X}_\\mathcal{L}=\\{(y_j, \\boldsymbol{x}_j), j=0\\ldots n-1\\}$. \n", + "\n", + "Let us assume that the true data is generated from a noisy model" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y}=f(\\boldsymbol{x}) + \\boldsymbol{\\epsilon}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\epsilon$ is normally distributed with mean zero and standard deviation $\\sigma^2$.\n", + "\n", + "In our derivation of the ordinary least squares method we defined then\n", + "an approximation to the function $f$ in terms of the parameters\n", + "$\\boldsymbol{\\beta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", + "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\beta}$. \n", + "\n", + "Thereafter we found the parameters $\\boldsymbol{\\beta}$ by optimizing the means squared error via the so-called cost function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{X},\\boldsymbol{\\beta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can rewrite this as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right]=\\frac{1}{n}\\sum_i(f_i-\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right])^2+\\frac{1}{n}\\sum_i(\\tilde{y}_i-\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right])^2+\\sigma^2.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The three terms represent the square of the bias of the learning\n", + "method, which can be thought of as the error caused by the simplifying\n", + "assumptions built into the method. The second term represents the\n", + "variance of the chosen model and finally the last terms is variance of\n", + "the error $\\boldsymbol{\\epsilon}$.\n", + "\n", + "To derive this equation, we need to recall that the variance of $\\boldsymbol{y}$ and $\\boldsymbol{\\epsilon}$ are both equal to $\\sigma^2$. The mean value of $\\boldsymbol{\\epsilon}$ is by definition equal to zero. Furthermore, the function $f$ is not a stochastics variable, idem for $\\boldsymbol{\\tilde{y}}$.\n", + "We use a more compact notation in terms of the expectation value" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right]=\\mathbb{E}\\left[(\\boldsymbol{f}+\\boldsymbol{\\epsilon}-\\boldsymbol{\\tilde{y}})^2\\right],\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and adding and subtracting $\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right]$ we get" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right]=\\mathbb{E}\\left[(\\boldsymbol{f}+\\boldsymbol{\\epsilon}-\\boldsymbol{\\tilde{y}}+\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right]-\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right])^2\\right],\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which, using the abovementioned expectation values can be rewritten as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right]=\\mathbb{E}\\left[(\\boldsymbol{y}-\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right])^2\\right]+\\mathrm{Var}\\left[\\boldsymbol{\\tilde{y}}\\right]+\\sigma^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "that is the rewriting in terms of the so-called bias, the variance of the model $\\boldsymbol{\\tilde{y}}$ and the variance of $\\boldsymbol{\\epsilon}$." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.pipeline import make_pipeline\n", + "from sklearn.utils import resample\n", + "\n", + "np.random.seed(2018)\n", + "\n", + "n = 500\n", + "n_boostraps = 100\n", + "degree = 18 # A quite high value, just to show.\n", + "noise = 0.1\n", + "\n", + "# Make data set.\n", + "x = np.linspace(-1, 3, n).reshape(-1, 1)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) + np.random.normal(0, 0.1, x.shape)\n", + "\n", + "# Hold out some test data that is never used in training.\n", + "x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2)\n", + "\n", + "# Combine x transformation and model into one operation.\n", + "# Not neccesary, but convenient.\n", + "model = make_pipeline(PolynomialFeatures(degree=degree), LinearRegression(fit_intercept=False))\n", + "\n", + "# The following (m x n_bootstraps) matrix holds the column vectors y_pred\n", + "# for each bootstrap iteration.\n", + "y_pred = np.empty((y_test.shape[0], n_boostraps))\n", + "for i in range(n_boostraps):\n", + " x_, y_ = resample(x_train, y_train)\n", + "\n", + " # Evaluate the new model on the same test data each time.\n", + " y_pred[:, i] = model.fit(x_, y_).predict(x_test).ravel()\n", + "\n", + "# Note: Expectations and variances taken w.r.t. different training\n", + "# data sets, hence the axis=1. Subsequent means are taken across the test data\n", + "# set in order to obtain a total value, but before this we have error/bias/variance\n", + "# calculated per data point in the test set.\n", + "# Note 2: The use of keepdims=True is important in the calculation of bias as this \n", + "# maintains the column vector form. Dropping this yields very unexpected results.\n", + "error = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) )\n", + "bias = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 )\n", + "variance = np.mean( np.var(y_pred, axis=1, keepdims=True) )\n", + "print('Error:', error)\n", + "print('Bias^2:', bias)\n", + "print('Var:', variance)\n", + "print('{} >= {} + {} = {}'.format(error, bias, variance, bias+variance))\n", + "\n", + "plt.plot(x[::5, :], y[::5, :], label='f(x)')\n", + "plt.scatter(x_test, y_test, label='Data points')\n", + "plt.scatter(x_test, np.mean(y_pred, axis=1), label='Pred')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.pipeline import make_pipeline\n", + "from sklearn.utils import resample\n", + "\n", + "np.random.seed(2018)\n", + "\n", + "n = 40\n", + "n_boostraps = 100\n", + "maxdegree = 14\n", + "\n", + "\n", + "# Make data set.\n", + "x = np.linspace(-3, 3, n).reshape(-1, 1)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)\n", + "error = np.zeros(maxdegree)\n", + "bias = np.zeros(maxdegree)\n", + "variance = np.zeros(maxdegree)\n", + "polydegree = np.zeros(maxdegree)\n", + "x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2)\n", + "\n", + "for degree in range(maxdegree):\n", + " model = make_pipeline(PolynomialFeatures(degree=degree), LinearRegression(fit_intercept=False))\n", + " y_pred = np.empty((y_test.shape[0], n_boostraps))\n", + " for i in range(n_boostraps):\n", + " x_, y_ = resample(x_train, y_train)\n", + " y_pred[:, i] = model.fit(x_, y_).predict(x_test).ravel()\n", + "\n", + " polydegree[degree] = degree\n", + " error[degree] = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) )\n", + " bias[degree] = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 )\n", + " variance[degree] = np.mean( np.var(y_pred, axis=1, keepdims=True) )\n", + " print('Polynomial degree:', degree)\n", + " print('Error:', error[degree])\n", + " print('Bias^2:', bias[degree])\n", + " print('Var:', variance[degree])\n", + " print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree]))\n", + "\n", + "plt.plot(polydegree, error, label='Error')\n", + "plt.plot(polydegree, bias, label='bias')\n", + "plt.plot(polydegree, variance, label='Variance')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The bias-variance tradeoff summarizes the fundamental tension in\n", + "machine learning, particularly supervised learning, between the\n", + "complexity of a model and the amount of training data needed to train\n", + "it. Since data is often limited, in practice it is often useful to\n", + "use a less-complex model with higher bias, that is a model whose asymptotic\n", + "performance is worse than another model because it is easier to\n", + "train and less sensitive to sampling noise arising from having a\n", + "finite-sized training dataset (smaller variance). \n", + "\n", + "\n", + "\n", + "The above equations tell us that in\n", + "order to minimize the expected test error, we need to select a\n", + "statistical learning method that simultaneously achieves low variance\n", + "and low bias. Note that variance is inherently a nonnegative quantity,\n", + "and squared bias is also nonnegative. Hence, we see that the expected\n", + "test MSE can never lie below $Var(\\epsilon)$, the irreducible error.\n", + "\n", + "\n", + "What do we mean by the variance and bias of a statistical learning\n", + "method? The variance refers to the amount by which our model would change if we\n", + "estimated it using a different training data set. Since the training\n", + "data are used to fit the statistical learning method, different\n", + "training data sets will result in a different estimate. But ideally the\n", + "estimate for our model should not vary too much between training\n", + "sets. However, if a method has high variance then small changes in\n", + "the training data can result in large changes in the model. In general, more\n", + "flexible statistical methods have higher variance.\n", + "\n", + "\n", + "You may also find this recent [article](https://www.pnas.org/content/116/32/15849) of interest." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\"\"\"\n", + "============================\n", + "Underfitting vs. Overfitting\n", + "============================\n", + "\n", + "This example demonstrates the problems of underfitting and overfitting and\n", + "how we can use linear regression with polynomial features to approximate\n", + "nonlinear functions. The plot shows the function that we want to approximate,\n", + "which is a part of the cosine function. In addition, the samples from the\n", + "real function and the approximations of different models are displayed. The\n", + "models have polynomial features of different degrees. We can see that a\n", + "linear function (polynomial with degree 1) is not sufficient to fit the\n", + "training samples. This is called **underfitting**. A polynomial of degree 4\n", + "approximates the true function almost perfectly. However, for higher degrees\n", + "the model will **overfit** the training data, i.e. it learns the noise of the\n", + "training data.\n", + "We evaluate quantitatively **overfitting** / **underfitting** by using\n", + "cross-validation. We calculate the mean squared error (MSE) on the validation\n", + "set, the higher, the less likely the model generalizes correctly from the\n", + "training data.\n", + "\"\"\"\n", + "\n", + "print(__doc__)\n", + "\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.pipeline import Pipeline\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "from sklearn.linear_model import LinearRegression\n", + "from sklearn.model_selection import cross_val_score\n", + "\n", + "\n", + "def true_fun(X):\n", + " return np.cos(1.5 * np.pi * X)\n", + "\n", + "np.random.seed(0)\n", + "\n", + "n_samples = 30\n", + "degrees = [1, 4, 15]\n", + "\n", + "X = np.sort(np.random.rand(n_samples))\n", + "y = true_fun(X) + np.random.randn(n_samples) * 0.1\n", + "\n", + "plt.figure(figsize=(14, 5))\n", + "for i in range(len(degrees)):\n", + " ax = plt.subplot(1, len(degrees), i + 1)\n", + " plt.setp(ax, xticks=(), yticks=())\n", + "\n", + " polynomial_features = PolynomialFeatures(degree=degrees[i],\n", + " include_bias=False)\n", + " linear_regression = LinearRegression()\n", + " pipeline = Pipeline([(\"polynomial_features\", polynomial_features),\n", + " (\"linear_regression\", linear_regression)])\n", + " pipeline.fit(X[:, np.newaxis], y)\n", + "\n", + " # Evaluate the models using crossvalidation\n", + " scores = cross_val_score(pipeline, X[:, np.newaxis], y,\n", + " scoring=\"neg_mean_squared_error\", cv=10)\n", + "\n", + " X_test = np.linspace(0, 1, 100)\n", + " plt.plot(X_test, pipeline.predict(X_test[:, np.newaxis]), label=\"Model\")\n", + " plt.plot(X_test, true_fun(X_test), label=\"True function\")\n", + " plt.scatter(X, y, edgecolor='b', s=20, label=\"Samples\")\n", + " plt.xlabel(\"x\")\n", + " plt.ylabel(\"y\")\n", + " plt.xlim((0, 1))\n", + " plt.ylim((-2, 2))\n", + " plt.legend(loc=\"best\")\n", + " plt.title(\"Degree {}\\nMSE = {:.2e}(+/- {:.2e})\".format(\n", + " degrees[i], -scores.mean(), scores.std()))\n", + "plt.show()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.utils import resample\n", + "from sklearn.metrics import mean_squared_error\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"EoS.csv\"),'r')\n", + "\n", + "# Read the EoS data as csv file and organize the data into two arrays with density and energies\n", + "EoS = pd.read_csv(infile, names=('Density', 'Energy'))\n", + "EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')\n", + "EoS = EoS.dropna()\n", + "Energies = EoS['Energy']\n", + "Density = EoS['Density']\n", + "# The design matrix now as function of various polytrops\n", + "\n", + "Maxpolydegree = 30\n", + "X = np.zeros((len(Density),Maxpolydegree))\n", + "X[:,0] = 1.0\n", + "testerror = np.zeros(Maxpolydegree)\n", + "trainingerror = np.zeros(Maxpolydegree)\n", + "polynomial = np.zeros(Maxpolydegree)\n", + "\n", + "trials = 100\n", + "for polydegree in range(1, Maxpolydegree):\n", + " polynomial[polydegree] = polydegree\n", + " for degree in range(polydegree):\n", + " X[:,degree] = Density**(degree/3.0)\n", + "\n", + "# loop over trials in order to estimate the expectation value of the MSE\n", + " testerror[polydegree] = 0.0\n", + " trainingerror[polydegree] = 0.0\n", + " for samples in range(trials):\n", + " x_train, x_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2)\n", + " model = LinearRegression(fit_intercept=True).fit(x_train, y_train)\n", + " ypred = model.predict(x_train)\n", + " ytilde = model.predict(x_test)\n", + " testerror[polydegree] += mean_squared_error(y_test, ytilde)\n", + " trainingerror[polydegree] += mean_squared_error(y_train, ypred) \n", + "\n", + " testerror[polydegree] /= trials\n", + " trainingerror[polydegree] /= trials\n", + " print(\"Degree of polynomial: %3d\"% polynomial[polydegree])\n", + " print(\"Mean squared error on training data: %.8f\" % trainingerror[polydegree])\n", + " print(\"Mean squared error on test data: %.8f\" % testerror[polydegree])\n", + "\n", + "plt.plot(polynomial, np.log10(trainingerror), label='Training Error')\n", + "plt.plot(polynomial, np.log10(testerror), label='Test Error')\n", + "plt.xlabel('Polynomial degree')\n", + "plt.ylabel('log10[MSE]')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.metrics import mean_squared_error\n", + "from sklearn.model_selection import KFold\n", + "from sklearn.model_selection import cross_val_score\n", + "\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"EoS.csv\"),'r')\n", + "\n", + "# Read the EoS data as csv file and organize the data into two arrays with density and energies\n", + "EoS = pd.read_csv(infile, names=('Density', 'Energy'))\n", + "EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')\n", + "EoS = EoS.dropna()\n", + "Energies = EoS['Energy']\n", + "Density = EoS['Density']\n", + "# The design matrix now as function of various polytrops\n", + "\n", + "Maxpolydegree = 30\n", + "X = np.zeros((len(Density),Maxpolydegree))\n", + "X[:,0] = 1.0\n", + "estimated_mse_sklearn = np.zeros(Maxpolydegree)\n", + "polynomial = np.zeros(Maxpolydegree)\n", + "k =5\n", + "kfold = KFold(n_splits = k)\n", + "\n", + "for polydegree in range(1, Maxpolydegree):\n", + " polynomial[polydegree] = polydegree\n", + " for degree in range(polydegree):\n", + " X[:,degree] = Density**(degree/3.0)\n", + " OLS = LinearRegression()\n", + "# loop over trials in order to estimate the expectation value of the MSE\n", + " estimated_mse_folds = cross_val_score(OLS, X, Energies, scoring='neg_mean_squared_error', cv=kfold)\n", + "#[:, np.newaxis]\n", + " estimated_mse_sklearn[polydegree] = np.mean(-estimated_mse_folds)\n", + "\n", + "plt.plot(polynomial, np.log10(estimated_mse_sklearn), label='Test Error')\n", + "plt.xlabel('Polynomial degree')\n", + "plt.ylabel('log10[MSE]')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import KFold\n", + "from sklearn.linear_model import Ridge\n", + "from sklearn.model_selection import cross_val_score\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "\n", + "# A seed just to ensure that the random numbers are the same for every run.\n", + "np.random.seed(3155)\n", + "# Generate the data.\n", + "n = 100\n", + "x = np.linspace(-3, 3, n).reshape(-1, 1)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)\n", + "# Decide degree on polynomial to fit\n", + "poly = PolynomialFeatures(degree = 10)\n", + "\n", + "# Decide which values of lambda to use\n", + "nlambdas = 500\n", + "lambdas = np.logspace(-3, 5, nlambdas)\n", + "# Initialize a KFold instance\n", + "k = 5\n", + "kfold = KFold(n_splits = k)\n", + "estimated_mse_sklearn = np.zeros(nlambdas)\n", + "i = 0\n", + "for lmb in lambdas:\n", + " ridge = Ridge(alpha = lmb)\n", + " estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)\n", + " estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)\n", + " i += 1\n", + "plt.figure()\n", + "plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')\n", + "plt.xlabel('log10(lambda)')\n", + "plt.ylabel('MSE')\n", + "plt.legend()\n", + "plt.show()" + ] + } + ], + "metadata": {}, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/doc/BookChapters/chapter3.ipynb b/doc/LectureNotes/_build/html/_sources/chapter3.ipynb similarity index 100% rename from doc/BookChapters/chapter3.ipynb rename to doc/LectureNotes/_build/html/_sources/chapter3.ipynb diff --git a/doc/LectureNotes/_build/html/_sources/chapter4.ipynb b/doc/LectureNotes/_build/html/_sources/chapter4.ipynb new file mode 100644 index 000000000..80fb450d4 --- /dev/null +++ b/doc/LectureNotes/_build/html/_sources/chapter4.ipynb @@ -0,0 +1,2875 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Logistic Regression\n", + "\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK3155/h20/forelesningsvideoer/LectureSeptember18.mp4?vrtx=view-as-webpage)\n", + "\n", + "\n", + "## Logistic Regression\n", + "\n", + "In linear regression our main interest was centered on learning the\n", + "coefficients of a functional fit (say a polynomial) in order to be\n", + "able to predict the response of a continuous variable on some unseen\n", + "data. The fit to the continuous variable $y_i$ is based on some\n", + "independent variables $\\hat{x}_i$. Linear regression resulted in\n", + "analytical expressions for standard ordinary Least Squares or Ridge\n", + "regression (in terms of matrices to invert) for several quantities,\n", + "ranging from the variance and thereby the confidence intervals of the\n", + "parameters $\\hat{\\beta}$ to the mean squared error. If we can invert\n", + "the product of the design matrices, linear regression gives then a\n", + "simple recipe for fitting our data.\n", + "\n", + "\n", + "Classification problems, however, are concerned with outcomes taking\n", + "the form of discrete variables (i.e. categories). We may for example,\n", + "on the basis of DNA sequencing for a number of patients, like to find\n", + "out which mutations are important for a certain disease; or based on\n", + "scans of various patients' brains, figure out if there is a tumor or\n", + "not; or given a specific physical system, we'd like to identify its\n", + "state, say whether it is an ordered or disordered system (typical\n", + "situation in solid state physics); or classify the status of a\n", + "patient, whether she/he has a stroke or not and many other similar\n", + "situations.\n", + "\n", + "The most common situation we encounter when we apply logistic\n", + "regression is that of two possible outcomes, normally denoted as a\n", + "binary outcome, true or false, positive or negative, success or\n", + "failure etc.\n", + "\n", + "\n", + "Logistic regression will also serve as our stepping stone towards\n", + "neural network algorithms and supervised deep learning. For logistic\n", + "learning, the minimization of the cost function leads to a non-linear\n", + "equation in the parameters $\\hat{\\beta}$. The optimization of the\n", + "problem calls therefore for minimization algorithms. This forms the\n", + "bottle neck of all machine learning algorithms, namely how to find\n", + "reliable minima of a multi-variable function. This leads us to the\n", + "family of gradient descent methods. The latter are the working horses\n", + "of basically all modern machine learning algorithms.\n", + "\n", + "We note also that many of the topics discussed here on logistic \n", + "regression are also commonly used in modern supervised Deep Learning\n", + "models, as we will see later.\n", + "\n", + "\n", + "\n", + "## Basics\n", + "\n", + "We consider the case where the dependent variables, also called the\n", + "responses or the outcomes, $y_i$ are discrete and only take values\n", + "from $k=0,\\dots,K-1$ (i.e. $K$ classes).\n", + "\n", + "The goal is to predict the\n", + "output classes from the design matrix $\\hat{X}\\in\\mathbb{R}^{n\\times p}$\n", + "made of $n$ samples, each of which carries $p$ features or predictors. The\n", + "primary goal is to identify the classes to which new unseen samples\n", + "belong.\n", + "\n", + "Let us specialize to the case of two classes only, with outputs\n", + "$y_i=0$ and $y_i=1$. Our outcomes could represent the status of a\n", + "credit card user that could default or not on her/his credit card\n", + "debt. That is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y_i = \\begin{bmatrix} 0 & \\mathrm{no}\\\\ 1 & \\mathrm{yes} \\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Before moving to the logistic model, let us try to use our linear\n", + "regression model to classify these two outcomes. We could for example\n", + "fit a linear model to the default case if $y_i > 0.5$ and the no\n", + "default case $y_i \\leq 0.5$.\n", + "\n", + "We would then have our \n", + "weighted linear combination, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "
\n", + "\n", + "$$\n", + "\\begin{equation}\n", + "\\hat{y} = \\hat{X}^T\\hat{\\beta} + \\hat{\\epsilon},\n", + "\\label{_auto1} \\tag{1}\n", + "\\end{equation}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\hat{y}$ is a vector representing the possible outcomes, $\\hat{X}$ is our\n", + "$n\\times p$ design matrix and $\\hat{\\beta}$ represents our estimators/predictors.\n", + "\n", + "\n", + "The main problem with our function is that it takes values on the\n", + "entire real axis. In the case of logistic regression, however, the\n", + "labels $y_i$ are discrete variables. A typical example is the credit\n", + "card data discussed below here, where we can set the state of\n", + "defaulting the debt to $y_i=1$ and not to $y_i=0$ for one the persons\n", + "in the data set (see the full example below).\n", + "\n", + "One simple way to get a discrete output is to have sign\n", + "functions that map the output of a linear regressor to values $\\{0,1\\}$,\n", + "$f(s_i)=sign(s_i)=1$ if $s_i\\ge 0$ and 0 if otherwise. \n", + "We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron\" model in the machine learning\n", + "literature. This model is extremely simple. However, in many cases it is more\n", + "favorable to use a ``soft\" classifier that outputs\n", + "the probability of a given category. This leads us to the logistic function.\n", + "\n", + "\n", + "The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "%matplotlib inline\n", + "\n", + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.utils import resample\n", + "from sklearn.metrics import mean_squared_error\n", + "from IPython.display import display\n", + "from pylab import plt, mpl\n", + "plt.style.use('seaborn')\n", + "mpl.rcParams['font.family'] = 'serif'\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"chddata.csv\"),'r')\n", + "\n", + "# Read the chd data as csv file and organize the data into arrays with age group, age, and chd\n", + "chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))\n", + "chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']\n", + "output = chd['CHD']\n", + "age = chd['Age']\n", + "agegroup = chd['Agegroup']\n", + "numberID = chd['ID'] \n", + "display(chd)\n", + "\n", + "plt.scatter(age, output, marker='o')\n", + "plt.axis([18,70.0,-0.1, 1.2])\n", + "plt.xlabel(r'Age')\n", + "plt.ylabel(r'CHD')\n", + "plt.title(r'Age distribution and Coronary heart disease')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "What we could attempt however is to plot the mean value for each group." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])\n", + "group = np.array([1, 2, 3, 4, 5, 6, 7, 8])\n", + "plt.plot(group, agegroupmean, \"r-\")\n", + "plt.axis([0,9,0, 1.0])\n", + "plt.xlabel(r'Age group')\n", + "plt.ylabel(r'CHD mean values')\n", + "plt.title(r'Mean values for each age group')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We are now trying to find a function $f(y\\vert x)$, that is a function which gives us an expected value for the output $y$ with a given input $x$.\n", + "In standard linear regression with a linear dependence on $x$, we would write this in terms of our model" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(y_i\\vert x_i)=\\beta_0+\\beta_1 x_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This expression implies however that $f(y_i\\vert x_i)$ could take any\n", + "value from minus infinity to plus infinity. If we however let\n", + "$f(y\\vert y)$ be represented by the mean value, the above example\n", + "shows us that we can constrain the function to take values between\n", + "zero and one, that is we have $0 \\le f(y_i\\vert x_i) \\le 1$. Looking\n", + "at our last curve we see also that it has an S-shaped form. This leads\n", + "us to a very popular model for the function $f$, namely the so-called\n", + "Sigmoid function or logistic model. We will consider this function as\n", + "representing the probability for finding a value of $y_i$ with a given\n", + "$x_i$.\n", + "\n", + "\n", + "## The logistic function\n", + "\n", + "Another widely studied model, is the so-called \n", + "perceptron model, which is an example of a \"hard classification\" model. We\n", + "will encounter this model when we discuss neural networks as\n", + "well. Each datapoint is deterministically assigned to a category (i.e\n", + "$y_i=0$ or $y_i=1$). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a \"soft\"\n", + "classifier that outputs the probability of a given category rather\n", + "than a single value. For example, given $x_i$, the classifier\n", + "outputs the probability of being in a category $k$. Logistic regression\n", + "is the most common example of a so-called soft classifier. In logistic\n", + "regression, the probability that a data point $x_i$\n", + "belongs to a category $y_i=\\{0,1\\}$ is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(t) = \\frac{1}{1+\\mathrm \\exp{-t}}=\\frac{\\exp{t}}{1+\\mathrm \\exp{t}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Note that $1-p(t)= p(-t)$.\n", + "\n", + "## Examples of likelihood functions used in logistic regression and nueral networks\n", + "\n", + "\n", + "The following code plots the logistic function, the step function and other functions we will encounter from here and on." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\"\"\"The sigmoid function (or the logistic curve) is a\n", + "function that takes any real number, z, and outputs a number (0,1).\n", + "It is useful in neural networks for assigning weights on a relative scale.\n", + "The value z is the weighted sum of parameters involved in the learning algorithm.\"\"\"\n", + "\n", + "import numpy\n", + "import matplotlib.pyplot as plt\n", + "import math as mt\n", + "\n", + "z = numpy.arange(-5, 5, .1)\n", + "sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))\n", + "sigma = sigma_fn(z)\n", + "\n", + "fig = plt.figure()\n", + "ax = fig.add_subplot(111)\n", + "ax.plot(z, sigma)\n", + "ax.set_ylim([-0.1, 1.1])\n", + "ax.set_xlim([-5,5])\n", + "ax.grid(True)\n", + "ax.set_xlabel('z')\n", + "ax.set_title('sigmoid function')\n", + "\n", + "plt.show()\n", + "\n", + "\"\"\"Step Function\"\"\"\n", + "z = numpy.arange(-5, 5, .02)\n", + "step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)\n", + "step = step_fn(z)\n", + "\n", + "fig = plt.figure()\n", + "ax = fig.add_subplot(111)\n", + "ax.plot(z, step)\n", + "ax.set_ylim([-0.5, 1.5])\n", + "ax.set_xlim([-5,5])\n", + "ax.grid(True)\n", + "ax.set_xlabel('z')\n", + "ax.set_title('step function')\n", + "\n", + "plt.show()\n", + "\n", + "\"\"\"tanh Function\"\"\"\n", + "z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)\n", + "t = numpy.tanh(z)\n", + "\n", + "fig = plt.figure()\n", + "ax = fig.add_subplot(111)\n", + "ax.plot(z, t)\n", + "ax.set_ylim([-1.0, 1.0])\n", + "ax.set_xlim([-2*mt.pi,2*mt.pi])\n", + "ax.grid(True)\n", + "ax.set_xlabel('z')\n", + "ax.set_title('tanh function')\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We assume now that we have two classes with $y_i$ either $0$ or $1$. Furthermore we assume also that we have only two parameters $\\beta$ in our fitting of the Sigmoid function, that is we define probabilities" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "p(y_i=1|x_i,\\hat{\\beta}) &= \\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}},\\nonumber\\\\\n", + "p(y_i=0|x_i,\\hat{\\beta}) &= 1 - p(y_i=1|x_i,\\hat{\\beta}),\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\hat{\\beta}$ are the weights we wish to extract from data, in our case $\\beta_0$ and $\\beta_1$. \n", + "\n", + "Note that we used" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(y_i=0\\vert x_i, \\hat{\\beta}) = 1-p(y_i=1\\vert x_i, \\hat{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In order to define the total likelihood for all possible outcomes from a \n", + "dataset $\\mathcal{D}=\\{(y_i,x_i)\\}$, with the binary labels\n", + "$y_i\\in\\{0,1\\}$ and where the data points are drawn independently, we use the so-called [Maximum Likelihood Estimation](https://en.wikipedia.org/wiki/Maximum_likelihood_estimation) (MLE) principle. \n", + "We aim thus at maximizing \n", + "the probability of seeing the observed data. We can then approximate the \n", + "likelihood in terms of the product of the individual probabilities of a specific outcome $y_i$, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "P(\\mathcal{D}|\\hat{\\beta})& = \\prod_{i=1}^n \\left[p(y_i=1|x_i,\\hat{\\beta})\\right]^{y_i}\\left[1-p(y_i=1|x_i,\\hat{\\beta}))\\right]^{1-y_i}\\nonumber \\\\\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "from which we obtain the log-likelihood and our **cost/loss** function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathcal{C}(\\hat{\\beta}) = \\sum_{i=1}^n \\left( y_i\\log{p(y_i=1|x_i,\\hat{\\beta})} + (1-y_i)\\log\\left[1-p(y_i=1|x_i,\\hat{\\beta}))\\right]\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Reordering the logarithms, we can rewrite the **cost/loss** function as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathcal{C}(\\hat{\\beta}) = \\sum_{i=1}^n \\left(y_i(\\beta_0+\\beta_1x_i) -\\log{(1+\\exp{(\\beta_0+\\beta_1x_i)})}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to $\\beta$.\n", + "Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathcal{C}(\\hat{\\beta})=-\\sum_{i=1}^n \\left(y_i(\\beta_0+\\beta_1x_i) -\\log{(1+\\exp{(\\beta_0+\\beta_1x_i)})}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This equation is known in statistics as the **cross entropy**. Finally, we note that just as in linear regression, \n", + "in practice we often supplement the cross-entropy with additional regularization terms, usually $L_1$ and $L_2$ regularization as we did for Ridge and Lasso regression.\n", + "\n", + "\n", + "The cross entropy is a convex function of the weights $\\hat{\\beta}$ and,\n", + "therefore, any local minimizer is a global minimizer. \n", + "\n", + "\n", + "Minimizing this\n", + "cost function with respect to the two parameters $\\beta_0$ and $\\beta_1$ we obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\hat{\\beta})}{\\partial \\beta_0} = -\\sum_{i=1}^n \\left(y_i -\\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}}\\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\hat{\\beta})}{\\partial \\beta_1} = -\\sum_{i=1}^n \\left(y_ix_i -x_i\\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Let us now define a vector $\\hat{y}$ with $n$ elements $y_i$, an\n", + "$n\\times p$ matrix $\\hat{X}$ which contains the $x_i$ values and a\n", + "vector $\\hat{p}$ of fitted probabilities $p(y_i\\vert x_i,\\hat{\\beta})$. We can rewrite in a more compact form the first\n", + "derivative of cost function as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\hat{\\beta})}{\\partial \\hat{\\beta}} = -\\hat{X}^T\\left(\\hat{y}-\\hat{p}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we in addition define a diagonal matrix $\\hat{W}$ with elements \n", + "$p(y_i\\vert x_i,\\hat{\\beta})(1-p(y_i\\vert x_i,\\hat{\\beta})$, we can obtain a compact expression of the second derivative as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial^2 \\mathcal{C}(\\hat{\\beta})}{\\partial \\hat{\\beta}\\partial \\hat{\\beta}^T} = \\hat{X}^T\\hat{W}\\hat{X}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with $p$ predictors" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{ \\frac{p(\\hat{\\beta}\\hat{x})}{1-p(\\hat{\\beta}\\hat{x})}} = \\beta_0+\\beta_1x_1+\\beta_2x_2+\\dots+\\beta_px_p.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here we defined $\\hat{x}=[1,x_1,x_2,\\dots,x_p]$ and $\\hat{\\beta}=[\\beta_0, \\beta_1, \\dots, \\beta_p]$ leading to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(\\hat{\\beta}\\hat{x})=\\frac{ \\exp{(\\beta_0+\\beta_1x_1+\\beta_2x_2+\\dots+\\beta_px_p)}}{1+\\exp{(\\beta_0+\\beta_1x_1+\\beta_2x_2+\\dots+\\beta_px_p)}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Till now we have mainly focused on two classes, the so-called binary\n", + "system. Suppose we wish to extend to $K$ classes. Let us for the sake\n", + "of simplicity assume we have only two predictors. We have then following model" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{\\frac{p(C=1\\vert x)}{p(K\\vert x)}} = \\beta_{10}+\\beta_{11}x_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{\\frac{p(C=2\\vert x)}{p(K\\vert x)}} = \\beta_{20}+\\beta_{21}x_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and so on till the class $C=K-1$ class" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{\\frac{p(C=K-1\\vert x)}{p(K\\vert x)}} = \\beta_{(K-1)0}+\\beta_{(K-1)1}x_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and the model is specified in term of $K-1$ so-called log-odds or\n", + "**logit** transformations.\n", + "\n", + "\n", + "\n", + "In our discussion of neural networks we will encounter the above again\n", + "in terms of a slightly modified function, the so-called **Softmax** function.\n", + "\n", + "The softmax function is used in various multiclass classification\n", + "methods, such as multinomial logistic regression (also known as\n", + "softmax regression), multiclass linear discriminant analysis, naive\n", + "Bayes classifiers, and artificial neural networks. Specifically, in\n", + "multinomial logistic regression and linear discriminant analysis, the\n", + "input to the function is the result of $K$ distinct linear functions,\n", + "and the predicted probability for the $k$-th class given a sample\n", + "vector $\\hat{x}$ and a weighting vector $\\hat{\\beta}$ is (with two\n", + "predictors):" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(C=k\\vert \\mathbf {x} )=\\frac{\\exp{(\\beta_{k0}+\\beta_{k1}x_1)}}{1+\\sum_{l=1}^{K-1}\\exp{(\\beta_{l0}+\\beta_{l1}x_1)}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "It is easy to extend to more predictors. The final class is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(C=K\\vert \\mathbf {x} )=\\frac{1}{1+\\sum_{l=1}^{K-1}\\exp{(\\beta_{l0}+\\beta_{l1}x_1)}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and they sum to one. Our earlier discussions were all specialized to\n", + "the case with two classes only. It is easy to see from the above that\n", + "what we derived earlier is compatible with these equations.\n", + "\n", + "To find the optimal parameters we would typically use a gradient\n", + "descent method. Newton's method and gradient descent methods are\n", + "discussed in the material on [optimization\n", + "methods](https://compphysics.github.io/MachineLearning/doc/pub/Splines/html/Splines-bs.html).\n", + "\n", + "## Wisconsin Cancer Data\n", + "\n", + "We show here how we can use a simple regression case on the breast\n", + "cancer data using Logistic regression as our algorithm for\n", + "classification." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.linear_model import LogisticRegression\n", + "\n", + "# Load the data\n", + "cancer = load_breast_cancer()\n", + "\n", + "X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)\n", + "print(X_train.shape)\n", + "print(X_test.shape)\n", + "# Logistic Regression\n", + "logreg = LogisticRegression(solver='lbfgs')\n", + "logreg.fit(X_train, y_train)\n", + "print(\"Test set accuracy with Logistic Regression: {:.2f}\".format(logreg.score(X_test,y_test)))\n", + "#now scale the data\n", + "from sklearn.preprocessing import StandardScaler\n", + "scaler = StandardScaler()\n", + "scaler.fit(X_train)\n", + "X_train_scaled = scaler.transform(X_train)\n", + "X_test_scaled = scaler.transform(X_test)\n", + "# Logistic Regression\n", + "logreg.fit(X_train_scaled, y_train)\n", + "print(\"Test set accuracy Logistic Regression with scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In addition to the above scores, we could also study the covariance (and the correlation matrix).\n", + "We use **Pandas** to compute the correlation matrix." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.linear_model import LogisticRegression\n", + "cancer = load_breast_cancer()\n", + "import pandas as pd\n", + "# Making a data frame\n", + "cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)\n", + "\n", + "fig, axes = plt.subplots(15,2,figsize=(10,20))\n", + "malignant = cancer.data[cancer.target == 0]\n", + "benign = cancer.data[cancer.target == 1]\n", + "ax = axes.ravel()\n", + "\n", + "for i in range(30):\n", + " _, bins = np.histogram(cancer.data[:,i], bins =50)\n", + " ax[i].hist(malignant[:,i], bins = bins, alpha = 0.5)\n", + " ax[i].hist(benign[:,i], bins = bins, alpha = 0.5)\n", + " ax[i].set_title(cancer.feature_names[i])\n", + " ax[i].set_yticks(())\n", + "ax[0].set_xlabel(\"Feature magnitude\")\n", + "ax[0].set_ylabel(\"Frequency\")\n", + "ax[0].legend([\"Malignant\", \"Benign\"], loc =\"best\")\n", + "fig.tight_layout()\n", + "plt.show()\n", + "\n", + "import seaborn as sns\n", + "correlation_matrix = cancerpd.corr().round(1)\n", + "# use the heatmap function from seaborn to plot the correlation matrix\n", + "# annot = True to print the values inside the square\n", + "plt.figure(figsize=(15,8))\n", + "sns.heatmap(data=correlation_matrix, annot=True)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the above example we note two things. In the first plot we display\n", + "the overlap of benign and malignant tumors as functions of the various\n", + "features in the Wisconsing breast cancer data set. We see that for\n", + "some of the features we can distinguish clearly the benign and\n", + "malignant cases while for other features we cannot. This can point to\n", + "us which features may be of greater interest when we wish to classify\n", + "a benign or not benign tumour.\n", + "\n", + "In the second figure we have computed the so-called correlation\n", + "matrix, which in our case with thirty features becomes a $30\\times 30$\n", + "matrix.\n", + "\n", + "We constructed this matrix using **pandas** via the statements" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and then" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "correlation_matrix = cancerpd.corr().round(1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Diagonalizing this matrix we can in turn say something about which\n", + "features are of relevance and which are not. This leads us to\n", + "the classical Principal Component Analysis (PCA) theorem with\n", + "applications. This will be discussed later this semester ([week 43](https://compphysics.github.io/MachineLearning/doc/pub/week43/html/week43-bs.html))." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.linear_model import LogisticRegression\n", + "\n", + "# Load the data\n", + "cancer = load_breast_cancer()\n", + "\n", + "X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)\n", + "print(X_train.shape)\n", + "print(X_test.shape)\n", + "# Logistic Regression\n", + "logreg = LogisticRegression(solver='lbfgs')\n", + "logreg.fit(X_train, y_train)\n", + "print(\"Test set accuracy with Logistic Regression: {:.2f}\".format(logreg.score(X_test,y_test)))\n", + "#now scale the data\n", + "from sklearn.preprocessing import StandardScaler\n", + "scaler = StandardScaler()\n", + "scaler.fit(X_train)\n", + "X_train_scaled = scaler.transform(X_train)\n", + "X_test_scaled = scaler.transform(X_test)\n", + "# Logistic Regression\n", + "logreg.fit(X_train_scaled, y_train)\n", + "print(\"Test set accuracy Logistic Regression with scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))\n", + "\n", + "\n", + "from sklearn.preprocessing import LabelEncoder\n", + "from sklearn.model_selection import cross_validate\n", + "#Cross validation\n", + "accuracy = cross_validate(logreg,X_test_scaled,y_test,cv=10)['test_score']\n", + "print(accuracy)\n", + "print(\"Test set accuracy with Logistic Regression and scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))\n", + "\n", + "\n", + "import scikitplot as skplt\n", + "y_pred = logreg.predict(X_test_scaled)\n", + "skplt.metrics.plot_confusion_matrix(y_test, y_pred, normalize=True)\n", + "plt.show()\n", + "y_probas = logreg.predict_proba(X_test_scaled)\n", + "skplt.metrics.plot_roc(y_test, y_probas)\n", + "plt.show()\n", + "skplt.metrics.plot_cumulative_gain(y_test, y_probas)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Optimization, the central part of any Machine Learning algortithm\n", + "\n", + "Almost every problem in machine learning and data science starts with\n", + "a dataset $X$, a model $g(\\beta)$, which is a function of the\n", + "parameters $\\beta$ and a cost function $C(X, g(\\beta))$ that allows\n", + "us to judge how well the model $g(\\beta)$ explains the observations\n", + "$X$. The model is fit by finding the values of $\\beta$ that minimize\n", + "the cost function. Ideally we would be able to solve for $\\beta$\n", + "analytically, however this is not possible in general and we must use\n", + "some approximative/numerical method to compute the minimum.\n", + "\n", + "\n", + "\n", + "## Revisiting our Logistic Regression case\n", + "\n", + "In our discussion on Logistic Regression we studied the \n", + "case of\n", + "two classes, with $y_i$ either\n", + "$0$ or $1$. Furthermore we assumed also that we have only two\n", + "parameters $\\beta$ in our fitting, that is we\n", + "defined probabilities" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "p(y_i=1|x_i,\\boldsymbol{\\beta}) &= \\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}},\\nonumber\\\\\n", + "p(y_i=0|x_i,\\boldsymbol{\\beta}) &= 1 - p(y_i=1|x_i,\\boldsymbol{\\beta}),\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\boldsymbol{\\beta}$ are the weights we wish to extract from data, in our case $\\beta_0$ and $\\beta_1$. \n", + "\n", + "\n", + "## The equations to solve\n", + "\n", + "Our compact equations used a definition of a vector $\\boldsymbol{y}$ with $n$\n", + "elements $y_i$, an $n\\times p$ matrix $\\boldsymbol{X}$ which contains the\n", + "$x_i$ values and a vector $\\boldsymbol{p}$ of fitted probabilities\n", + "$p(y_i\\vert x_i,\\boldsymbol{\\beta})$. We rewrote in a more compact form\n", + "the first derivative of the cost function as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = -\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{p}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we in addition define a diagonal matrix $\\boldsymbol{W}$ with elements \n", + "$p(y_i\\vert x_i,\\boldsymbol{\\beta})(1-p(y_i\\vert x_i,\\boldsymbol{\\beta})$, we can obtain a compact expression of the second derivative as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial^2 \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}\\partial \\boldsymbol{\\beta}^T} = \\boldsymbol{X}^T\\boldsymbol{W}\\boldsymbol{X}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This defines what is called the Hessian matrix.\n", + "\n", + "\n", + "## Solving using Newton-Raphson's method\n", + "\n", + "If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives. \n", + "\n", + "Our iterative scheme is then given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{new}} = \\boldsymbol{\\beta}^{\\mathrm{old}}-\\left(\\frac{\\partial^2 \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}\\partial \\boldsymbol{\\beta}^T}\\right)^{-1}_{\\boldsymbol{\\beta}^{\\mathrm{old}}}\\times \\left(\\frac{\\partial \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}}\\right)_{\\boldsymbol{\\beta}^{\\mathrm{old}}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or in matrix form as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{new}} = \\boldsymbol{\\beta}^{\\mathrm{old}}-\\left(\\boldsymbol{X}^T\\boldsymbol{W}\\boldsymbol{X} \\right)^{-1}\\times \\left(-\\boldsymbol{X}^T(\\boldsymbol{y}-\\boldsymbol{p}) \\right)_{\\boldsymbol{\\beta}^{\\mathrm{old}}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The right-hand side is computed with the old values of $\\beta$. \n", + "\n", + "If we can compute these matrices, in particular the Hessian, the above is often the easiest method to implement. \n", + "\n", + "\n", + "\n", + "## Brief reminder on Newton-Raphson's method\n", + "\n", + "Let us quickly remind ourselves how we derive the above method.\n", + "\n", + "Perhaps the most celebrated of all one-dimensional root-finding\n", + "routines is Newton's method, also called the Newton-Raphson\n", + "method. This method requires the evaluation of both the\n", + "function $f$ and its derivative $f'$ at arbitrary points. \n", + "If you can only calculate the derivative\n", + "numerically and/or your function is not of the smooth type, we\n", + "normally discourage the use of this method.\n", + "\n", + "\n", + "## The equations\n", + "\n", + "The Newton-Raphson formula consists geometrically of extending the\n", + "tangent line at a current point until it crosses zero, then setting\n", + "the next guess to the abscissa of that zero-crossing. The mathematics\n", + "behind this method is rather simple. Employing a Taylor expansion for\n", + "$x$ sufficiently close to the solution $s$, we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "
\n", + "\n", + "$$\n", + "f(s)=0=f(x)+(s-x)f'(x)+\\frac{(s-x)^2}{2}f''(x) +\\dots.\n", + " \\label{eq:taylornr} \\tag{2}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "For small enough values of the function and for well-behaved\n", + "functions, the terms beyond linear are unimportant, hence we obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(x)+(s-x)f'(x)\\approx 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "yielding" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "s\\approx x-\\frac{f(x)}{f'(x)}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Having in mind an iterative procedure, it is natural to start iterating with" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "x_{n+1}=x_n-\\frac{f(x_n)}{f'(x_n)}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Simple geometric interpretation\n", + "\n", + "The above is Newton-Raphson's method. It has a simple geometric\n", + "interpretation, namely $x_{n+1}$ is the point where the tangent from\n", + "$(x_n,f(x_n))$ crosses the $x$-axis. Close to the solution,\n", + "Newton-Raphson converges fast to the desired result. However, if we\n", + "are far from a root, where the higher-order terms in the series are\n", + "important, the Newton-Raphson formula can give grossly inaccurate\n", + "results. For instance, the initial guess for the root might be so far\n", + "from the true root as to let the search interval include a local\n", + "maximum or minimum of the function. If an iteration places a trial\n", + "guess near such a local extremum, so that the first derivative nearly\n", + "vanishes, then Newton-Raphson may fail totally\n", + "\n", + "\n", + "\n", + "## Extending to more than one variable\n", + "\n", + "Newton's method can be generalized to systems of several non-linear equations\n", + "and variables. Consider the case with two equations" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{array}{cc} f_1(x_1,x_2) &=0\\\\\n", + " f_2(x_1,x_2) &=0,\\end{array}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which we Taylor expand to obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{array}{cc} 0=f_1(x_1+h_1,x_2+h_2)=&f_1(x_1,x_2)+h_1\n", + " \\partial f_1/\\partial x_1+h_2\n", + " \\partial f_1/\\partial x_2+\\dots\\\\\n", + " 0=f_2(x_1+h_1,x_2+h_2)=&f_2(x_1,x_2)+h_1\n", + " \\partial f_2/\\partial x_1+h_2\n", + " \\partial f_2/\\partial x_2+\\dots\n", + " \\end{array}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Defining the Jacobian matrix ${\\bf \\boldsymbol{J}}$ we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\bf \\boldsymbol{J}}=\\left( \\begin{array}{cc}\n", + " \\partial f_1/\\partial x_1 & \\partial f_1/\\partial x_2 \\\\\n", + " \\partial f_2/\\partial x_1 &\\partial f_2/\\partial x_2\n", + " \\end{array} \\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we can rephrase Newton's method as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\left(\\begin{array}{c} x_1^{n+1} \\\\ x_2^{n+1} \\end{array} \\right)=\n", + "\\left(\\begin{array}{c} x_1^{n} \\\\ x_2^{n} \\end{array} \\right)+\n", + "\\left(\\begin{array}{c} h_1^{n} \\\\ h_2^{n} \\end{array} \\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have defined" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\left(\\begin{array}{c} h_1^{n} \\\\ h_2^{n} \\end{array} \\right)=\n", + " -{\\bf \\boldsymbol{J}}^{-1}\n", + " \\left(\\begin{array}{c} f_1(x_1^{n},x_2^{n}) \\\\ f_2(x_1^{n},x_2^{n}) \\end{array} \\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We need thus to compute the inverse of the Jacobian matrix and it\n", + "is to understand that difficulties may\n", + "arise in case ${\\bf \\boldsymbol{J}}$ is nearly singular.\n", + "\n", + "It is rather straightforward to extend the above scheme to systems of\n", + "more than two non-linear equations. In our case, the Jacobian matrix is given by the Hessian that represents the second derivative of cost function. \n", + "\n", + "\n", + "\n", + "\n", + "## Steepest descent\n", + "\n", + "The basic idea of gradient descent is\n", + "that a function $F(\\mathbf{x})$, \n", + "$\\mathbf{x} \\equiv (x_1,\\cdots,x_n)$, decreases fastest if one goes from $\\bf {x}$ in the\n", + "direction of the negative gradient $-\\nabla F(\\mathbf{x})$.\n", + "\n", + "It can be shown that if" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{x}_{k+1} = \\mathbf{x}_k - \\gamma_k \\nabla F(\\mathbf{x}_k),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\gamma_k > 0$.\n", + "\n", + "For $\\gamma_k$ small enough, then $F(\\mathbf{x}_{k+1}) \\leq\n", + "F(\\mathbf{x}_k)$. This means that for a sufficiently small $\\gamma_k$\n", + "we are always moving towards smaller function values, i.e a minimum.\n", + "\n", + "\n", + "## More on Steepest descent\n", + "\n", + "The previous observation is the basis of the method of steepest\n", + "descent, which is also referred to as just gradient descent (GD). One\n", + "starts with an initial guess $\\mathbf{x}_0$ for a minimum of $F$ and\n", + "computes new approximations according to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{x}_{k+1} = \\mathbf{x}_k - \\gamma_k \\nabla F(\\mathbf{x}_k), \\ \\ k \\geq 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The parameter $\\gamma_k$ is often referred to as the step length or\n", + "the learning rate within the context of Machine Learning.\n", + "\n", + "\n", + "## The ideal\n", + "\n", + "Ideally the sequence $\\{\\mathbf{x}_k \\}_{k=0}$ converges to a global\n", + "minimum of the function $F$. In general we do not know if we are in a\n", + "global or local minimum. In the special case when $F$ is a convex\n", + "function, all local minima are also global minima, so in this case\n", + "gradient descent can converge to the global solution. The advantage of\n", + "this scheme is that it is conceptually simple and straightforward to\n", + "implement. However the method in this form has some severe\n", + "limitations:\n", + "\n", + "In machine learing we are often faced with non-convex high dimensional\n", + "cost functions with many local minima. Since GD is deterministic we\n", + "will get stuck in a local minimum, if the method converges, unless we\n", + "have a very good intial guess. This also implies that the scheme is\n", + "sensitive to the chosen initial condition.\n", + "\n", + "Note that the gradient is a function of $\\mathbf{x} =\n", + "(x_1,\\cdots,x_n)$ which makes it expensive to compute numerically.\n", + "\n", + "\n", + "\n", + "## The sensitiveness of the gradient descent\n", + "\n", + "The gradient descent method \n", + "is sensitive to the choice of learning rate $\\gamma_k$. This is due\n", + "to the fact that we are only guaranteed that $F(\\mathbf{x}_{k+1}) \\leq\n", + "F(\\mathbf{x}_k)$ for sufficiently small $\\gamma_k$. The problem is to\n", + "determine an optimal learning rate. If the learning rate is chosen too\n", + "small the method will take a long time to converge and if it is too\n", + "large we can experience erratic behavior.\n", + "\n", + "Many of these shortcomings can be alleviated by introducing\n", + "randomness. One such method is that of Stochastic Gradient Descent\n", + "(SGD), see below.\n", + "\n", + "\n", + "\n", + "## Convex functions\n", + "\n", + "Ideally we want our cost/loss function to be convex(concave).\n", + "\n", + "First we give the definition of a convex set: A set $C$ in\n", + "$\\mathbb{R}^n$ is said to be convex if, for all $x$ and $y$ in $C$ and\n", + "all $t \\in (0,1)$ , the point $(1 − t)x + ty$ also belongs to\n", + "C. Geometrically this means that every point on the line segment\n", + "connecting $x$ and $y$ is in $C$ as discussed below.\n", + "\n", + "The convex subsets of $\\mathbb{R}$ are the intervals of\n", + "$\\mathbb{R}$. Examples of convex sets of $\\mathbb{R}^2$ are the\n", + "regular polygons (triangles, rectangles, pentagons, etc...).\n", + "\n", + "\n", + "## Convex function\n", + "\n", + "**Convex function**: Let $X \\subset \\mathbb{R}^n$ be a convex set. Assume that the function $f: X \\rightarrow \\mathbb{R}$ is continuous, then $f$ is said to be convex if $$f(tx_1 + (1-t)x_2) \\leq tf(x_1) + (1-t)f(x_2) $$ for all $x_1, x_2 \\in X$ and for all $t \\in [0,1]$. If $\\leq$ is replaced with a strict inequaltiy in the definition, we demand $x_1 \\neq x_2$ and $t\\in(0,1)$ then $f$ is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting $f(x_1)$ and $f(x_2)$, the value of the function on the interval $[x_1,x_2]$ is always below the line as illustrated below.\n", + "\n", + "\n", + "## Conditions on convex functions\n", + "\n", + "In the following we state first and second-order conditions which\n", + "ensures convexity of a function $f$. We write $D_f$ to denote the\n", + "domain of $f$, i.e the subset of $R^n$ where $f$ is defined. For more\n", + "details and proofs we refer to: [S. Boyd and L. Vandenberghe. Convex Optimization. Cambridge University Press](http://stanford.edu/boyd/cvxbook/, 2004).\n", + "\n", + "**First order condition.**\n", + "\n", + "Suppose $f$ is differentiable (i.e $\\nabla f(x)$ is well defined for\n", + "all $x$ in the domain of $f$). Then $f$ is convex if and only if $D_f$\n", + "is a convex set and $$f(y) \\geq f(x) + \\nabla f(x)^T (y-x) $$ holds\n", + "for all $x,y \\in D_f$. This condition means that for a convex function\n", + "the first order Taylor expansion (right hand side above) at any point\n", + "a global under estimator of the function. To convince yourself you can\n", + "make a drawing of $f(x) = x^2+1$ and draw the tangent line to $f(x)$ and\n", + "note that it is always below the graph.\n", + "\n", + "\n", + "\n", + "**Second order condition.**\n", + "\n", + "Assume that $f$ is twice\n", + "differentiable, i.e the Hessian matrix exists at each point in\n", + "$D_f$. Then $f$ is convex if and only if $D_f$ is a convex set and its\n", + "Hessian is positive semi-definite for all $x\\in D_f$. For a\n", + "single-variable function this reduces to $f''(x) \\geq 0$. Geometrically this means that $f$ has nonnegative curvature\n", + "everywhere.\n", + "\n", + "\n", + "\n", + "This condition is particularly useful since it gives us an procedure for determining if the function under consideration is convex, apart from using the definition.\n", + "\n", + "\n", + "## More on convex functions\n", + "\n", + "The next result is of great importance to us and the reason why we are\n", + "going on about convex functions. In machine learning we frequently\n", + "have to minimize a loss/cost function in order to find the best\n", + "parameters for the model we are considering. \n", + "\n", + "Ideally we want the\n", + "global minimum (for high-dimensional models it is hard to know\n", + "if we have local or global minimum). However, if the cost/loss function\n", + "is convex the following result provides invaluable information:\n", + "\n", + "**Any minimum is global for convex functions.**\n", + "\n", + "Consider the problem of finding $x \\in \\mathbb{R}^n$ such that $f(x)$\n", + "is minimal, where $f$ is convex and differentiable. Then, any point\n", + "$x^*$ that satisfies $\\nabla f(x^*) = 0$ is a global minimum.\n", + "\n", + "\n", + "\n", + "This result means that if we know that the cost/loss function is convex and we are able to find a minimum, we are guaranteed that it is a global minimum.\n", + "\n", + "\n", + "## Some simple problems\n", + "\n", + "1. Show that $f(x)=x^2$ is convex for $x \\in \\mathbb{R}$ using the definition of convexity. Hint: If you re-write the definition, $f$ is convex if the following holds for all $x,y \\in D_f$ and any $\\lambda \\in [0,1]$ $\\lambda f(x)+(1-\\lambda)f(y)-f(\\lambda x + (1-\\lambda) y ) \\geq 0$.\n", + "\n", + "2. Using the second order condition show that the following functions are convex on the specified domain.\n", + "\n", + " * $f(x) = e^x$ is convex for $x \\in \\mathbb{R}$.\n", + "\n", + " * $g(x) = -\\ln(x)$ is convex for $x \\in (0,\\infty)$.\n", + "\n", + "\n", + "3. Let $f(x) = x^2$ and $g(x) = e^x$. Show that $f(g(x))$ and $g(f(x))$ is convex for $x \\in \\mathbb{R}$. Also show that if $f(x)$ is any convex function than $h(x) = e^{f(x)}$ is convex.\n", + "\n", + "4. A norm is any function that satisfy the following properties\n", + "\n", + " * $f(\\alpha x) = |\\alpha| f(x)$ for all $\\alpha \\in \\mathbb{R}$.\n", + "\n", + " * $f(x+y) \\leq f(x) + f(y)$\n", + "\n", + " * $f(x) \\leq 0$ for all $x \\in \\mathbb{R}^n$ with equality if and only if $x = 0$\n", + "\n", + "\n", + "Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this).\n", + "\n", + "\n", + "\n", + "## Friday September 25\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember25.mp4?vrtx=view-as-webpage) and [link to handwritten notes](https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/NotesSeptember25.pdf).\n", + "\n", + "\n", + "\n", + "## Standard steepest descent\n", + "\n", + "\n", + "Before we proceed, we would like to discuss the approach called the\n", + "**standard Steepest descent** (different from the above steepest descent discussion), which again leads to us having to be able\n", + "to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG).\n", + "\n", + "[The success of the CG method](https://www.cs.cmu.edu/~quake-papers/painless-conjugate-gradient.pdf)\n", + "for finding solutions of non-linear problems is based on the theory\n", + "of conjugate gradients for linear systems of equations. It belongs to\n", + "the class of iterative methods for solving problems from linear\n", + "algebra of the type" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{x} = \\boldsymbol{b}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the iterative process we end up with a problem like" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}= \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\boldsymbol{r}$ is the so-called residual or error in the iterative process.\n", + "\n", + "When we have found the exact solution, $\\boldsymbol{r}=0$.\n", + "\n", + "\n", + "## Gradient method\n", + "\n", + "The residual is zero when we reach the minimum of the quadratic equation" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "P(\\boldsymbol{x})=\\frac{1}{2}\\boldsymbol{x}^T\\boldsymbol{A}\\boldsymbol{x} - \\boldsymbol{x}^T\\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with the constraint that the matrix $\\boldsymbol{A}$ is positive definite and\n", + "symmetric. This defines also the Hessian and we want it to be positive definite. \n", + "\n", + "\n", + "\n", + "## Steepest descent method\n", + "\n", + "We denote the initial guess for $\\boldsymbol{x}$ as $\\boldsymbol{x}_0$. \n", + "We can assume without loss of generality that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_0=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or consider the system" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{z} = \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "instead.\n", + "\n", + "\n", + "\n", + "## Steepest descent method\n", + "One can show that the solution $\\boldsymbol{x}$ is also the unique minimizer of the quadratic form" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(\\boldsymbol{x}) = \\frac{1}{2}\\boldsymbol{x}^T\\boldsymbol{A}\\boldsymbol{x} - \\boldsymbol{x}^T \\boldsymbol{x} , \\quad \\boldsymbol{x}\\in\\mathbf{R}^n.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This suggests taking the first basis vector $\\boldsymbol{r}_1$ (see below for definition) \n", + "to be the gradient of $f$ at $\\boldsymbol{x}=\\boldsymbol{x}_0$, \n", + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{x}_0-\\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and \n", + "$\\boldsymbol{x}_0=0$ it is equal $-\\boldsymbol{b}$.\n", + "\n", + "\n", + "\n", + "\n", + "## Final expressions\n", + "We can compute the residual iteratively as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_{k+1}=\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_{k+1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{b}-\\boldsymbol{A}(\\boldsymbol{x}_k+\\alpha_k\\boldsymbol{r}_k),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_k)-\\alpha_k\\boldsymbol{A}\\boldsymbol{r}_k,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which gives" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\alpha_k = \\frac{\\boldsymbol{r}_k^T\\boldsymbol{r}_k}{\\boldsymbol{r}_k^T\\boldsymbol{A}\\boldsymbol{r}_k}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "leading to the iterative scheme" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_{k+1}=\\boldsymbol{x}_k-\\alpha_k\\boldsymbol{r}_{k},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Steepest descent example" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import numpy.linalg as la\n", + "\n", + "import scipy.optimize as sopt\n", + "\n", + "import matplotlib.pyplot as pt\n", + "from mpl_toolkits.mplot3d import axes3d\n", + "\n", + "def f(x):\n", + " return 0.5*x[0]**2 + 2.5*x[1]**2\n", + "\n", + "def df(x):\n", + " return np.array([x[0], 5*x[1]])\n", + "\n", + "fig = pt.figure()\n", + "ax = fig.gca(projection=\"3d\")\n", + "\n", + "xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]\n", + "fmesh = f(np.array([xmesh, ymesh]))\n", + "ax.plot_surface(xmesh, ymesh, fmesh)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "And then as countor plot" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "pt.axis(\"equal\")\n", + "pt.contour(xmesh, ymesh, fmesh)\n", + "guesses = [np.array([2, 2./5])]" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Find guesses" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "x = guesses[-1]\n", + "s = -df(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Run it!" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "def f1d(alpha):\n", + " return f(x + alpha*s)\n", + "\n", + "alpha_opt = sopt.golden(f1d)\n", + "next_guess = x + alpha_opt * s\n", + "guesses.append(next_guess)\n", + "print(next_guess)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "What happened?" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "pt.axis(\"equal\")\n", + "pt.contour(xmesh, ymesh, fmesh, 50)\n", + "it_array = np.array(guesses)\n", + "pt.plot(it_array.T[0], it_array.T[1], \"x-\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method\n", + "In the CG method we define so-called conjugate directions and two vectors \n", + "$\\boldsymbol{s}$ and $\\boldsymbol{t}$\n", + "are said to be\n", + "conjugate if" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{s}^T\\boldsymbol{A}\\boldsymbol{t}= 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The philosophy of the CG method is to perform searches in various conjugate directions\n", + "of our vectors $\\boldsymbol{x}_i$ obeying the above criterion, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_i^T\\boldsymbol{A}\\boldsymbol{x}_j= 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Two vectors are conjugate if they are orthogonal with respect to \n", + "this inner product. Being conjugate is a symmetric relation: if $\\boldsymbol{s}$ is conjugate to $\\boldsymbol{t}$, then $\\boldsymbol{t}$ is conjugate to $\\boldsymbol{s}$.\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "An example is given by the eigenvectors of the matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{v}_i^T\\boldsymbol{A}\\boldsymbol{v}_j= \\lambda\\boldsymbol{v}_i^T\\boldsymbol{v}_j,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which is zero unless $i=j$.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "Assume now that we have a symmetric positive-definite matrix $\\boldsymbol{A}$ of size\n", + "$n\\times n$. At each iteration $i+1$ we obtain the conjugate direction of a vector" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_{i+1}=\\boldsymbol{x}_{i}+\\alpha_i\\boldsymbol{p}_{i}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We assume that $\\boldsymbol{p}_{i}$ is a sequence of $n$ mutually conjugate directions. \n", + "Then the $\\boldsymbol{p}_{i}$ form a basis of $R^n$ and we can expand the solution \n", + "$ \\boldsymbol{A}\\boldsymbol{x} = \\boldsymbol{b}$ in this basis, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x} = \\sum^{n}_{i=1} \\alpha_i \\boldsymbol{p}_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method\n", + "The coefficients are given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{A}\\mathbf{x} = \\sum^{n}_{i=1} \\alpha_i \\mathbf{A} \\mathbf{p}_i = \\mathbf{b}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Multiplying with $\\boldsymbol{p}_k^T$ from the left gives" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{p}_k^T \\boldsymbol{A}\\boldsymbol{x} = \\sum^{n}_{i=1} \\alpha_i\\boldsymbol{p}_k^T \\boldsymbol{A}\\boldsymbol{p}_i= \\boldsymbol{p}_k^T \\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and we can define the coefficients $\\alpha_k$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\alpha_k = \\frac{\\boldsymbol{p}_k^T \\boldsymbol{b}}{\\boldsymbol{p}_k^T \\boldsymbol{A} \\boldsymbol{p}_k}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method and iterations\n", + "\n", + "If we choose the conjugate vectors $\\boldsymbol{p}_k$ carefully, \n", + "then we may not need all of them to obtain a good approximation to the solution \n", + "$\\boldsymbol{x}$. \n", + "We want to regard the conjugate gradient method as an iterative method. \n", + "This will us to solve systems where $n$ is so large that the direct \n", + "method would take too much time.\n", + "\n", + "We denote the initial guess for $\\boldsymbol{x}$ as $\\boldsymbol{x}_0$. \n", + "We can assume without loss of generality that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_0=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or consider the system" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{z} = \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "instead.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "One can show that the solution $\\boldsymbol{x}$ is also the unique minimizer of the quadratic form" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(\\boldsymbol{x}) = \\frac{1}{2}\\boldsymbol{x}^T\\boldsymbol{A}\\boldsymbol{x} - \\boldsymbol{x}^T \\boldsymbol{x} , \\quad \\boldsymbol{x}\\in\\mathbf{R}^n.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This suggests taking the first basis vector $\\boldsymbol{p}_1$ \n", + "to be the gradient of $f$ at $\\boldsymbol{x}=\\boldsymbol{x}_0$, \n", + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{x}_0-\\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and \n", + "$\\boldsymbol{x}_0=0$ it is equal $-\\boldsymbol{b}$.\n", + "The other vectors in the basis will be conjugate to the gradient, \n", + "hence the name conjugate gradient method.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "Let $\\boldsymbol{r}_k$ be the residual at the $k$-th step:" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_k=\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_k.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Note that $\\boldsymbol{r}_k$ is the negative gradient of $f$ at \n", + "$\\boldsymbol{x}=\\boldsymbol{x}_k$, \n", + "so the gradient descent method would be to move in the direction $\\boldsymbol{r}_k$. \n", + "Here, we insist that the directions $\\boldsymbol{p}_k$ are conjugate to each other, \n", + "so we take the direction closest to the gradient $\\boldsymbol{r}_k$ \n", + "under the conjugacy constraint. \n", + "This gives the following expression" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{p}_{k+1}=\\boldsymbol{r}_k-\\frac{\\boldsymbol{p}_k^T \\boldsymbol{A}\\boldsymbol{r}_k}{\\boldsymbol{p}_k^T\\boldsymbol{A}\\boldsymbol{p}_k} \\boldsymbol{p}_k.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method\n", + "We can also compute the residual iteratively as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_{k+1}=\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_{k+1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{b}-\\boldsymbol{A}(\\boldsymbol{x}_k+\\alpha_k\\boldsymbol{p}_k),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_k)-\\alpha_k\\boldsymbol{A}\\boldsymbol{p}_k,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which gives" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_{k+1}=\\boldsymbol{r}_k-\\boldsymbol{A}\\boldsymbol{p}_{k},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Revisiting our first homework\n", + "\n", + "We will use linear regression as a case study for the gradient descent\n", + "methods. Linear regression is a great test case for the gradient\n", + "descent methods discussed in the lectures since it has several\n", + "desirable properties such as:\n", + "\n", + "1. An analytical solution (recall homework set 1).\n", + "\n", + "2. The gradient can be computed analytically.\n", + "\n", + "3. The cost function is convex which guarantees that gradient descent converges for small enough learning rates\n", + "\n", + "We revisit an example similar to what we had in the first homework set. We had a function of the type" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "x = 2*np.random.rand(m,1)\n", + "y = 4+3*x+np.random.randn(m,1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $x_i \\in [0,1] $ is chosen randomly using a uniform distribution. Additionally we have a stochastic noise chosen according to a normal distribution $\\cal {N}(0,1)$. \n", + "The linear regression model is given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "h_\\beta(x) = \\boldsymbol{y} = \\beta_0 + \\beta_1 x,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "such that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y}_i = \\beta_0 + \\beta_1 x_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Gradient descent example\n", + "\n", + "Let $\\mathbf{y} = (y_1,\\cdots,y_n)^T$, $\\mathbf{\\boldsymbol{y}} = (\\boldsymbol{y}_1,\\cdots,\\boldsymbol{y}_n)^T$ and $\\beta = (\\beta_0, \\beta_1)^T$\n", + "\n", + "It is convenient to write $\\mathbf{\\boldsymbol{y}} = X\\beta$ where $X \\in \\mathbb{R}^{100 \\times 2} $ is the design matrix given by (we keep the intercept here)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "X \\equiv \\begin{bmatrix}\n", + "1 & x_1 \\\\\n", + "\\vdots & \\vdots \\\\\n", + "1 & x_{100} & \\\\\n", + "\\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The cost/loss/risk function is given by (" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\beta) = \\frac{1}{n}||X\\beta-\\mathbf{y}||_{2}^{2} = \\frac{1}{n}\\sum_{i=1}^{100}\\left[ (\\beta_0 + \\beta_1 x_i)^2 - 2 y_i (\\beta_0 + \\beta_1 x_i) + y_i^2\\right]\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and we want to find $\\beta$ such that $C(\\beta)$ is minimized.\n", + "\n", + "\n", + "## The derivative of the cost/loss function\n", + "\n", + "Computing $\\partial C(\\beta) / \\partial \\beta_0$ and $\\partial C(\\beta) / \\partial \\beta_1$ we can show that the gradient can be written as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_{\\beta} C(\\beta) = \\frac{2}{n}\\begin{bmatrix} \\sum_{i=1}^{100} \\left(\\beta_0+\\beta_1x_i-y_i\\right) \\\\\n", + "\\sum_{i=1}^{100}\\left( x_i (\\beta_0+\\beta_1x_i)-y_ix_i\\right) \\\\\n", + "\\end{bmatrix} = \\frac{2}{n}X^T(X\\beta - \\mathbf{y}),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $X$ is the design matrix defined above.\n", + "\n", + "\n", + "## The Hessian matrix\n", + "The Hessian matrix of $C(\\beta)$ is given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{H} \\equiv \\begin{bmatrix}\n", + "\\frac{\\partial^2 C(\\beta)}{\\partial \\beta_0^2} & \\frac{\\partial^2 C(\\beta)}{\\partial \\beta_0 \\partial \\beta_1} \\\\\n", + "\\frac{\\partial^2 C(\\beta)}{\\partial \\beta_0 \\partial \\beta_1} & \\frac{\\partial^2 C(\\beta)}{\\partial \\beta_1^2} & \\\\\n", + "\\end{bmatrix} = \\frac{2}{n}X^T X.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This result implies that $C(\\beta)$ is a convex function since the matrix $X^T X$ always is positive semi-definite.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Simple program\n", + "\n", + "We can now write a program that minimizes $C(\\beta)$ using the gradient descent method with a constant learning rate $\\gamma$ according to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_{k+1} = \\beta_k - \\gamma \\nabla_\\beta C(\\beta_k), \\ k=0,1,\\cdots\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can use the expression we computed for the gradient and let use a\n", + "$\\beta_0$ be chosen randomly and let $\\gamma = 0.001$. Stop iterating\n", + "when $||\\nabla_\\beta C(\\beta_k) || \\leq \\epsilon = 10^{-8}$. **Note that the code below does not include the latter stop criterion**.\n", + "\n", + "And finally we can compare our solution for $\\beta$ with the analytic result given by \n", + "$\\beta= (X^TX)^{-1} X^T \\mathbf{y}$.\n", + "\n", + "\n", + "## Gradient Descent Example\n", + "\n", + "Here our simple example" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\n", + "# Importing various packages\n", + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from mpl_toolkits.mplot3d import Axes3D\n", + "from matplotlib import cm\n", + "from matplotlib.ticker import LinearLocator, FormatStrFormatter\n", + "import sys\n", + "\n", + "# the number of datapoints\n", + "n = 100\n", + "x = 2*np.random.rand(n,1)\n", + "y = 4+3*x+np.random.randn(n,1)\n", + "\n", + "X = np.c_[np.ones((n,1)), x]\n", + "# Hessian matrix\n", + "H = (2.0/n)* X.T @ X\n", + "# Get the eigenvalues\n", + "EigValues, EigVectors = np.linalg.eig(H)\n", + "print(EigValues)\n", + "\n", + "beta_linreg = np.linalg.inv(X.T @ X) @ X.T @ y\n", + "print(beta_linreg)\n", + "beta = np.random.randn(2,1)\n", + "\n", + "eta = 1.0/np.max(EigValues)\n", + "Niterations = 1000\n", + "\n", + "for iter in range(Niterations):\n", + " gradient = (2.0/n)*X.T @ (X @ beta-y)\n", + " beta -= eta*gradient\n", + "\n", + "print(beta)\n", + "xnew = np.array([[0],[2]])\n", + "xbnew = np.c_[np.ones((2,1)), xnew]\n", + "ypredict = xbnew.dot(beta)\n", + "ypredict2 = xbnew.dot(beta_linreg)\n", + "plt.plot(xnew, ypredict, \"r-\")\n", + "plt.plot(xnew, ypredict2, \"b-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,2.0,0, 15.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Gradient descent example')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## And a corresponding example using **scikit-learn**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import SGDRegressor\n", + "\n", + "n = 100\n", + "x = 2*np.random.rand(n,1)\n", + "y = 4+3*x+np.random.randn(n,1)\n", + "\n", + "X = np.c_[np.ones((n,1)), x]\n", + "beta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)\n", + "print(beta_linreg)\n", + "sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1)\n", + "sgdreg.fit(x,y.ravel())\n", + "print(sgdreg.intercept_, sgdreg.coef_)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Gradient descent and Ridge\n", + "\n", + "We have also discussed Ridge regression where the loss function contains a regularized term given by the $L_2$ norm of $\\beta$," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C_{\\text{ridge}}(\\beta) = \\frac{1}{n}||X\\beta -\\mathbf{y}||^2 + \\lambda ||\\beta||^2, \\ \\lambda \\geq 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In order to minimize $C_{\\text{ridge}}(\\beta)$ using GD we only have adjust the gradient as follows" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_\\beta C_{\\text{ridge}}(\\beta) = \\frac{2}{n}\\begin{bmatrix} \\sum_{i=1}^{100} \\left(\\beta_0+\\beta_1x_i-y_i\\right) \\\\\n", + "\\sum_{i=1}^{100}\\left( x_i (\\beta_0+\\beta_1x_i)-y_ix_i\\right) \\\\\n", + "\\end{bmatrix} + 2\\lambda\\begin{bmatrix} \\beta_0 \\\\ \\beta_1\\end{bmatrix} = 2 (X^T(X\\beta - \\mathbf{y})+\\lambda \\beta).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can easily extend our program to minimize $C_{\\text{ridge}}(\\beta)$ using gradient descent and compare with the analytical solution given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_{\\text{ridge}} = \\left(X^T X + \\lambda I_{2 \\times 2} \\right)^{-1} X^T \\mathbf{y}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Program example for gradient descent with Ridge Regression" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from mpl_toolkits.mplot3d import Axes3D\n", + "from matplotlib import cm\n", + "from matplotlib.ticker import LinearLocator, FormatStrFormatter\n", + "import sys\n", + "\n", + "# the number of datapoints\n", + "n = 100\n", + "x = 2*np.random.rand(n,1)\n", + "y = 4+3*x+np.random.randn(n,1)\n", + "\n", + "X = np.c_[np.ones((n,1)), x]\n", + "XT_X = X.T @ X\n", + "\n", + "#Ridge parameter lambda\n", + "lmbda = 0.001\n", + "Id = lmbda* np.eye(XT_X.shape[0])\n", + "\n", + "beta_linreg = np.linalg.inv(XT_X+Id) @ X.T @ y\n", + "print(beta_linreg)\n", + "# Start plain gradient descent\n", + "beta = np.random.randn(2,1)\n", + "\n", + "eta = 0.1\n", + "Niterations = 100\n", + "\n", + "for iter in range(Niterations):\n", + " gradients = 2.0/n*X.T @ (X @ (beta)-y)+2*lmbda*beta\n", + " beta -= eta*gradients\n", + "\n", + "print(beta)\n", + "ypredict = X @ beta\n", + "ypredict2 = X @ beta_linreg\n", + "plt.plot(x, ypredict, \"r-\")\n", + "plt.plot(x, ypredict2, \"b-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,2.0,0, 15.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Gradient descent example for Ridge')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Using gradient descent methods, limitations\n", + "\n", + "* **Gradient descent (GD) finds local minima of our function**. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our cost/loss/risk function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.\n", + "\n", + "* **GD is sensitive to initial conditions**. One consequence of the local nature of GD is that initial conditions matter. Depending on where one starts, one will end up at a different local minima. Therefore, it is very important to think about how one initializes the training process. This is true for GD as well as more complicated variants of GD.\n", + "\n", + "* **Gradients are computationally expensive to calculate for large datasets**. In many cases in statistics and ML, the cost/loss/risk function is a sum of terms, with one term for each data point. For example, in linear regression, $E \\propto \\sum_{i=1}^n (y_i - \\mathbf{w}^T\\cdot\\mathbf{x}_i)^2$; for logistic regression, the square error is replaced by the cross entropy. To calculate the gradient we have to sum over *all* $n$ data points. Doing this at every GD step becomes extremely computationally expensive. An ingenious solution to this, is to calculate the gradients using small subsets of the data called \"mini batches\". This has the added benefit of introducing stochasticity into our algorithm.\n", + "\n", + "* **GD is very sensitive to choices of learning rates**. GD is extremely sensitive to the choice of learning rates. If the learning rate is very small, the training process take an extremely long time. For larger learning rates, GD can diverge and give poor results. Furthermore, depending on what the local landscape looks like, we have to modify the learning rates to ensure convergence. Ideally, we would *adaptively* choose the learning rates to match the landscape.\n", + "\n", + "* **GD treats all directions in parameter space uniformly.** Another major drawback of GD is that unlike Newton's method, the learning rate for GD is the same in all directions in parameter space. For this reason, the maximum learning rate is set by the behavior of the steepest direction and this can significantly slow down training. Ideally, we would like to take large steps in flat directions and small steps in steep directions. Since we are exploring rugged landscapes where curvatures change, this requires us to keep track of not only the gradient but second derivatives. The ideal scenario would be to calculate the Hessian but this proves to be too computationally expensive. \n", + "\n", + "* GD can take exponential time to escape saddle points, even with random initialization. As we mentioned, GD is extremely sensitive to initial condition since it determines the particular local minimum GD would eventually reach. However, even with a good initialization scheme, through the introduction of randomness, GD can still take exponential time to escape saddle points.\n", + "\n", + "## Stochastic Gradient Descent\n", + "\n", + "Stochastic gradient descent (SGD) and variants thereof address some of\n", + "the shortcomings of the Gradient descent method discussed above.\n", + "\n", + "The underlying idea of SGD comes from the observation that the cost\n", + "function, which we want to minimize, can almost always be written as a\n", + "sum over $n$ data points $\\{\\mathbf{x}_i\\}_{i=1}^n$," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\mathbf{\\beta}) = \\sum_{i=1}^n c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Computation of gradients\n", + "\n", + "This in turn means that the gradient can be\n", + "computed as a sum over $i$-gradients" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_\\beta C(\\mathbf{\\beta}) = \\sum_i^n \\nabla_\\beta c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Stochasticity/randomness is introduced by only taking the\n", + "gradient on a subset of the data called minibatches. If there are $n$\n", + "data points and the size of each minibatch is $M$, there will be $n/M$\n", + "minibatches. We denote these minibatches by $B_k$ where\n", + "$k=1,\\cdots,n/M$.\n", + "\n", + "\n", + "## SGD example\n", + "As an example, suppose we have $10$ data points $(\\mathbf{x}_1,\\cdots, \\mathbf{x}_{10})$ \n", + "and we choose to have $M=5$ minibathces,\n", + "then each minibatch contains two data points. In particular we have\n", + "$B_1 = (\\mathbf{x}_1,\\mathbf{x}_2), \\cdots, B_5 =\n", + "(\\mathbf{x}_9,\\mathbf{x}_{10})$. Note that if you choose $M=1$ you\n", + "have only a single batch with all data points and on the other extreme,\n", + "you may choose $M=n$ resulting in a minibatch for each datapoint, i.e\n", + "$B_k = \\mathbf{x}_k$.\n", + "\n", + "The idea is now to approximate the gradient by replacing the sum over\n", + "all data points with a sum over the data points in one the minibatches\n", + "picked at random in each gradient descent step" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_{\\beta}\n", + "C(\\mathbf{\\beta}) = \\sum_{i=1}^n \\nabla_\\beta c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta}) \\rightarrow \\sum_{i \\in B_k}^n \\nabla_\\beta\n", + "c_i(\\mathbf{x}_i, \\mathbf{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## The gradient step\n", + "\n", + "Thus a gradient descent step now looks like" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_{j+1} = \\beta_j - \\gamma_j \\sum_{i \\in B_k}^n \\nabla_\\beta c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta})\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $k$ is picked at random with equal\n", + "probability from $[1,n/M]$. An iteration over the number of\n", + "minibathces (n/M) is commonly referred to as an epoch. Thus it is\n", + "typical to choose a number of epochs and for each epoch iterate over\n", + "the number of minibatches, as exemplified in the code below.\n", + "\n", + "\n", + "## Simple example code" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np \n", + "\n", + "n = 100 #100 datapoints \n", + "M = 5 #size of each minibatch\n", + "m = int(n/M) #number of minibatches\n", + "n_epochs = 10 #number of epochs\n", + "\n", + "j = 0\n", + "for epoch in range(1,n_epochs+1):\n", + " for i in range(m):\n", + " k = np.random.randint(m) #Pick the k-th minibatch at random\n", + " #Compute the gradient using the data in minibatch Bk\n", + " #Compute new suggestion for \n", + " j += 1" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Taking the gradient only on a subset of the data has two important\n", + "benefits. First, it introduces randomness which decreases the chance\n", + "that our opmization scheme gets stuck in a local minima. Second, if\n", + "the size of the minibatches are small relative to the number of\n", + "datapoints ($M < n$), the computation of the gradient is much\n", + "cheaper since we sum over the datapoints in the $k-th$ minibatch and not\n", + "all $n$ datapoints.\n", + "\n", + "\n", + "## When do we stop?\n", + "\n", + "A natural question is when do we stop the search for a new minimum?\n", + "One possibility is to compute the full gradient after a given number\n", + "of epochs and check if the norm of the gradient is smaller than some\n", + "threshold and stop if true. However, the condition that the gradient\n", + "is zero is valid also for local minima, so this would only tell us\n", + "that we are close to a local/global minimum. However, we could also\n", + "evaluate the cost function at this point, store the result and\n", + "continue the search. If the test kicks in at a later stage we can\n", + "compare the values of the cost function and keep the $\\beta$ that\n", + "gave the lowest value.\n", + "\n", + "\n", + "## Slightly different approach\n", + "\n", + "Another approach is to let the step length $\\gamma_j$ depend on the\n", + "number of epochs in such a way that it becomes very small after a\n", + "reasonable time such that we do not move at all.\n", + "\n", + "As an example, let $e = 0,1,2,3,\\cdots$ denote the current epoch and let $t_0, t_1 > 0$ be two fixed numbers. Furthermore, let $t = e \\cdot m + i$ where $m$ is the number of minibatches and $i=0,\\cdots,m-1$. Then the function $$\\gamma_j(t; t_0, t_1) = \\frac{t_0}{t+t_1} $$ goes to zero as the number of epochs gets large. I.e. we start with a step length $\\gamma_j (0; t_0, t_1) = t_0/t_1$ which decays in *time* $t$.\n", + "\n", + "In this way we can fix the number of epochs, compute $\\beta$ and\n", + "evaluate the cost function at the end. Repeating the computation will\n", + "give a different result since the scheme is random by design. Then we\n", + "pick the final $\\beta$ that gives the lowest value of the cost\n", + "function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np \n", + "\n", + "def step_length(t,t0,t1):\n", + " return t0/(t+t1)\n", + "\n", + "n = 100 #100 datapoints \n", + "M = 5 #size of each minibatch\n", + "m = int(n/M) #number of minibatches\n", + "n_epochs = 500 #number of epochs\n", + "t0 = 1.0\n", + "t1 = 10\n", + "\n", + "gamma_j = t0/t1\n", + "j = 0\n", + "for epoch in range(1,n_epochs+1):\n", + " for i in range(m):\n", + " k = np.random.randint(m) #Pick the k-th minibatch at random\n", + " #Compute the gradient using the data in minibatch Bk\n", + " #Compute new suggestion for beta\n", + " t = epoch*m+i\n", + " gamma_j = step_length(t,t0,t1)\n", + " j += 1\n", + "\n", + "print(\"gamma_j after %d epochs: %g\" % (n_epochs,gamma_j))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Program for stochastic gradient" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "from math import exp, sqrt\n", + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import SGDRegressor\n", + "\n", + "m = 100\n", + "x = 2*np.random.rand(m,1)\n", + "y = 4+3*x+np.random.randn(m,1)\n", + "\n", + "X = np.c_[np.ones((m,1)), x]\n", + "theta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)\n", + "print(\"Own inversion\")\n", + "print(theta_linreg)\n", + "sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1)\n", + "sgdreg.fit(x,y.ravel())\n", + "print(\"sgdreg from scikit\")\n", + "print(sgdreg.intercept_, sgdreg.coef_)\n", + "\n", + "\n", + "theta = np.random.randn(2,1)\n", + "eta = 0.1\n", + "Niterations = 1000\n", + "\n", + "\n", + "for iter in range(Niterations):\n", + " gradients = 2.0/m*X.T @ ((X @ theta)-y)\n", + " theta -= eta*gradients\n", + "print(\"theta from own gd\")\n", + "print(theta)\n", + "\n", + "xnew = np.array([[0],[2]])\n", + "Xnew = np.c_[np.ones((2,1)), xnew]\n", + "ypredict = Xnew.dot(theta)\n", + "ypredict2 = Xnew.dot(theta_linreg)\n", + "\n", + "\n", + "n_epochs = 50\n", + "t0, t1 = 5, 50\n", + "def learning_schedule(t):\n", + " return t0/(t+t1)\n", + "\n", + "theta = np.random.randn(2,1)\n", + "\n", + "for epoch in range(n_epochs):\n", + " for i in range(m):\n", + " random_index = np.random.randint(m)\n", + " xi = X[random_index:random_index+1]\n", + " yi = y[random_index:random_index+1]\n", + " gradients = 2 * xi.T @ ((xi @ theta)-yi)\n", + " eta = learning_schedule(epoch*m+i)\n", + " theta = theta - eta*gradients\n", + "print(\"theta from own sdg\")\n", + "print(theta)\n", + "\n", + "plt.plot(xnew, ypredict, \"r-\")\n", + "plt.plot(xnew, ypredict2, \"b-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,2.0,0, 15.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Random numbers ')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "**Challenge**: try to write a similar code for a Logistic Regression case." + ] + } + ], + "metadata": {}, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/doc/LectureNotes/_build/html/_sources/content.md b/doc/LectureNotes/_build/html/_sources/content.md new file mode 100644 index 000000000..0f6aca77a --- /dev/null +++ b/doc/LectureNotes/_build/html/_sources/content.md @@ -0,0 +1,5 @@ +Content in Jupyter Book +======================= + +There are many ways to write content in Jupyter Book. This short section +covers a few tips for how to do so. diff --git a/doc/LectureNotes/_build/html/_sources/intro.md b/doc/LectureNotes/_build/html/_sources/intro.md new file mode 100644 index 000000000..811e85085 --- /dev/null +++ b/doc/LectureNotes/_build/html/_sources/intro.md @@ -0,0 +1,145 @@ +# Applied Data Analysis and Machine Learning + + +## Introduction + +Probability theory and statistical methods play a central role in science. Nowadays we are +surrounded by huge amounts of data. For example, there are about one trillion web pages; more than one +hour of video is uploaded to YouTube every second, amounting to years of content every +day; the genomes of 1000s of people, each of which has a length of more than a billion base pairs, have +been sequenced by various labs and so on. This deluge of data calls for automated methods of data analysis, +which is exactly what machine learning aims at providing. + +## Learning outcomes + +This course aims at giving you insights and knowledge about many of the central algorithms used in Data Analysis and Machine Learning. The course is project based and through various numerical projects, normally three, you will be exposed to fundamental research problems in these fields, with the aim to reproduce state of the art scientific results. Both supervised and unsupervised methods will be covered. The emphasis is on a frequentist approach, although we will try to link it with a Bayesian approach as well. You will learn to develop and structure large codes for studying different cases where Machine Learning is applied to, get acquainted with computing facilities and learn to handle large scientific projects. A good scientific and ethical conduct is emphasized throughout the course. More specifically, after this course you will + +- Learn about basic data analysis, statistical analysis, Bayesian statistics, Monte Carlo sampling, data optimization and machine learning; +- Be capable of extending the acquired knowledge to other systems and cases; +- Have an understanding of central algorithms used in data analysis and machine learning; +- Understand linear methods for regression and classification, from ordinary least squares, via Lasso and Ridge to Logistic regression; +- Learn about neural networks and deep learning methods for supervised and unsupervised learning. Emphasis on feed forward neural networks, convolutional and recurrent neural networks; +- Learn about about decision trees, random forests, bagging and boosting methods; +- Learn about support vector machines and kernel transformations; +- Reduction of data sets, from PCA to clustering; +- Autoencoders and Reinforcement Learning; +- Work on numerical projects to illustrate the theory. The projects play a central role and you are expected to know modern programming languages like Python or C++ and/or Fortran (Fortran2003 or later). + +## Prerequisites + +Basic knowledge in programming and mathematics, with an emphasis on +linear algebra. Knowledge of Python or/and C++ as programming +languages is strongly recommended and experience with Jupiter notebook +is recommended. Required courses are the equivalents to the University +of Oslo mathematics courses MAT1100, MAT1110, MAT1120 and at least one +of the corresponding computing and programming courses INF1000/INF1110 +or MAT-INF1100/MAT-INF1100L/BIOS1100/KJM-INF1100. Most universities +offer nowadays a basic programming course (often compulsory) where +Python is the recurring programming language. + + +## The course has two central parts + +1. Statistical analysis and optimization of data +2. Machine learning + +These topics will be scattered thorughout the course and may not necessarily be taught separately. Rather, we will often take an approach (during the lectures and project/exercise sessions) where say elements from statistical data analysis are mixed with specific Machine Learning algorithms. + +### Statistical analysis and optimization of data + +The following topics will be covered +- Basic concepts, expectation values, variance, covariance, correlation functions and errors; +- Simpler models, binomial distribution, the Poisson distribution, simple and multivariate normal distributions; +- Central elements of Bayesian statistics and modeling; +- Gradient methods for data optimization, +- Monte Carlo methods, Markov chains, Gibbs sampling and Metropolis-Hastings sampling; +- Estimation of errors and resampling techniques such as the cross-validation, blocking, bootstrapping and jackknife methods; +- Principal Component Analysis (PCA) and its mathematical foundation + +### Machine learning + +The following topics will be covered: +- Linear Regression and Logistic Regression; +- Neural networks and deep learning, including convolutional and recurrent neural networks +- Decisions trees, Random Forests, Bagging and Boosting +- Support vector machines +- Bayesian linear and logistic regression +- Boltzmann Machines +- Unsupervised learning Dimensionality reduction, from PCA to cluster models + +Hands-on demonstrations, exercises and projects aim at deepening your understanding of these topics. + +Computational aspects play a central role and you are +expected to work on numerical examples and projects which illustrate +the theory and varous algorithms discussed during the lectures. We recommend strongly to form small project groups of 2-3 participants, if possible. + +## Required Technologies + +Course participants are expected to have their own laptops/PCs. We use _Git_ as version control software and the usage of providers like _GitHub_, _GitLab_ or similar are strongly recommended. + +We will make extensive use of Python as programming language and its +myriad of available libraries. You will find +Jupyter notebooks invaluable in your work. You can run _R_ +codes in the Jupyter/IPython notebooks, with the immediate benefit of +visualizing your data. You can also use compiled languages like C++, +Rust, Julia, Fortran etc if you prefer. The focus in these lectures will be mainly +on Python. + + +If you have Python installed and you feel +pretty familiar with installing different packages, we recommend that +you install the following Python packages via _pip_ as + +* pip install numpy scipy matplotlib ipython scikit-learn mglearn sympy pandas pillow + +For OSX users we recommend, after having installed Xcode, to +install _brew_. Brew allows for a seamless installation of additional +software via for example + +* brew install python3 + +For Linux users, with its variety of distributions like for example the widely popular Ubuntu distribution, +you can use _pip_ as well and simply install Python as + +* sudo apt-get install python3 + +### Python installers + +If you don't want to perform these operations separately and venture +into the hassle of exploring how to set up dependencies and paths, we +recommend two widely used distrubutions which set up all relevant +dependencies for Python, namely + +* Anaconda:https://docs.anaconda.com/, + +which is an open source +distribution of the Python and R programming languages for large-scale +data processing, predictive analytics, and scientific computing, that +aims to simplify package management and deployment. Package versions +are managed by the package management system _conda_. + +* Enthought canopy:https://www.enthought.com/product/canopy/ + +is a Python +distribution for scientific and analytic computing distribution and +analysis environment, available for free and under a commercial +license. + +Furthermore, Google's Colab:https://colab.research.google.com/notebooks/welcome.ipynb is a free Jupyter notebook environment that requires +no setup and runs entirely in the cloud. Try it out! + +### Useful Python libraries +Here we list several useful Python libraries we strongly recommend (if you use anaconda many of these are already there) + +* _NumPy_:https://www.numpy.org/ is a highly popular library for large, multi-dimensional arrays and matrices, along with a large collection of high-level mathematical functions to operate on these arrays +* _The pandas_:https://pandas.pydata.org/ library provides high-performance, easy-to-use data structures and data analysis tools +* _Xarray_:http://xarray.pydata.org/en/stable/ is a Python package that makes working with labelled multi-dimensional arrays simple, efficient, and fun! +* _Scipy_:https://www.scipy.org/ (pronounced “Sigh Pie”) is a Python-based ecosystem of open-source software for mathematics, science, and engineering. +* _Matplotlib_:https://matplotlib.org/ is a Python 2D plotting library which produces publication quality figures in a variety of hardcopy formats and interactive environments across platforms. +* _Autograd_:https://github.com/HIPS/autograd can automatically differentiate native Python and Numpy code. It can handle a large subset of Python's features, including loops, ifs, recursion and closures, and it can even take derivatives of derivatives of derivatives +* _SymPy_:https://www.sympy.org/en/index.html is a Python library for symbolic mathematics. +* _scikit-learn_:https://scikit-learn.org/stable/ has simple and efficient tools for machine learning, data mining and data analysis +* _TensorFlow_:https://www.tensorflow.org/ is a Python library for fast numerical computing created and released by Google +* _Keras_:https://keras.io/ is a high-level neural networks API, written in Python and capable of running on top of TensorFlow, CNTK, or Theano +* And many more such as _pytorch_:https://pytorch.org/, _Theano_:https://pypi.org/project/Theano/ etc + diff --git a/doc/LectureNotes/_build/html/_sources/schedule.md b/doc/LectureNotes/_build/html/_sources/schedule.md new file mode 100644 index 000000000..13527dc67 --- /dev/null +++ b/doc/LectureNotes/_build/html/_sources/schedule.md @@ -0,0 +1,14 @@ +# Teaching schedule with links to material + + +This course will be delivered in a hybrid mode, with online lectures and on site or online laboratory sessions. + +1. Four lectures per week, Fall semester, 10 ECTS. The lectures will be fully online. The lectures will be recorded and linked to this site and the official University of Oslo website for the course; +2. Two hours of laboratory sessions for work on computational projects and exercises for each group. Due to social distancing, at most 15 participants can attend. There will also be fully digital laboratory sessions for those who cannot attend; +3. Three projects which are graded and count 1/3 each of the final grade; +4. A selected number of weekly assignments; +5. The course is part of the CS Master of Science program, but is open to other bachelor and Master of Science students at the University of Oslo; +6. The course is offered as a FYS-MAT4155 (Master of Science level) and a FYS-MAT3155 (senior undergraduate) course; +7. Videos of teaching material are available via the links at https://compphysics.github.io/MachineLearning/doc/web/course.html; +8. Weekly emails with summary of activities will be mailed to all participants; + diff --git a/doc/LectureNotes/_build/html/_sources/teachers.md b/doc/LectureNotes/_build/html/_sources/teachers.md new file mode 100644 index 000000000..4b934a68b --- /dev/null +++ b/doc/LectureNotes/_build/html/_sources/teachers.md @@ -0,0 +1,23 @@ +# Teachers and Grading + + +## Instructor information +* _Name_: Morten Hjorth-Jensen +* _Email_: morten.hjorth-jensen@fys.uio.no +* _Phone_: +47-48257387 +* _Office_: Department of Physics, University of Oslo, Eastern wing, room FØ470 +* _Office hours_: *Anytime*! In Fall Semester 2020 (FS20), as a rule of thumb office hours are planned via computer or telephone. Individual or group office hours will be performed via zoom. Feel free to send an email for planning. In person meetings may also be possible if allowed by the University of Oslo's COVID-19 instructions (see below for links). + + +## Grading +Grading scale: Grades are awarded on a scale from A to F, where A is the best grade and F is a fail. There are three projects which are graded and each project counts 1/3 of the final grade. The total score is thus the average from all three projects. + +The final number of points is based on the average of all projects (including eventual additional points) and the grade follows the following table: + + * 92-100 points: A + * 77-91 points: B + * 58-76 points: C + * 46-57 points: D + * 40-45 points: E + * 0-39 points: F-failed + diff --git a/doc/LectureNotes/_build/html/_sources/textbooks.md b/doc/LectureNotes/_build/html/_sources/textbooks.md new file mode 100644 index 000000000..92d980b58 --- /dev/null +++ b/doc/LectureNotes/_build/html/_sources/textbooks.md @@ -0,0 +1,38 @@ +# Textbooks + + +_Recommended textbooks_: +- Christopher M. Bishop, Pattern Recognition and Machine Learning, Springer, https://www.springer.com/gp/book/9780387310732. This is the main textbook and this course covers chapters 1-7, 11 and 12. +- Trevor Hastie, Robert Tibshirani, Jerome H. Friedman, The Elements of Statistical Learning, Springer, https://www.springer.com/gp/book/9780387848570. This is a well-known text and serves as additional text. +- Aurelien Geron, Hands‑On Machine Learning with Scikit‑Learn and TensorFlow, O'Reilly, https://www.oreilly.com/library/view/hands-on-machine-learning/9781492032632/. This text is very useful since it contains many code examples. + +The books by Bishop and Hastie et al. can be downloaded for free if you access the university library via an IP number of your home university. + + + +_General learning book on statistical analysis_: +- Christian Robert and George Casella, Monte Carlo Statistical Methods, Springer +- Peter Hoff, A first course in Bayesian statistical models, Springer + +_General Machine Learning Books_: +- Kevin Murphy, Machine Learning: A Probabilistic Perspective, MIT Press +- Christopher M. Bishop, Pattern Recognition and Machine Learning, Springer +- David J.C. MacKay, Information Theory, Inference, and Learning Algorithms, Cambridge University Press +- David Barber, Bayesian Reasoning and Machine Learning, Cambridge University Press + +## Links to relevant courses at the University of Oslo +The link here https://www.mn.uio.no/english/research/about/centre-focus/innovation/data-science/studies/ gives an excellent overview of courses on Machine learning at UiO. + +- _STK2100 Machine learning and statistical methods for prediction and classification_ http://www.uio.no/studier/emner/matnat/math/STK2100/index-eng.html. +- _IN3050 Introduction to Artificial Intelligence and Machine Learning_ https://www.uio.no/studier/emner/matnat/ifi/IN3050/index-eng.html. Introductory course in machine learning and AI with an algorithmic approach. +- _STK-INF3000/4000 Selected Topics in Data Science_ http://www.uio.no/studier/emner/matnat/math/STK-INF3000/index-eng.html. The course provides insight into selected contemporary relevant topics within Data Science. +- _IN4080 Natural Language Processing_ https://www.uio.no/studier/emner/matnat/ifi/IN4080/index.html. Probabilistic and machine learning techniques applied to natural language processing. +- _STK-IN4300 Statistical learning methods in Data Science_ https://www.uio.no/studier/emner/matnat/math/STK-IN4300/index-eng.html. An advanced introduction to statistical and machine learning. For students with a good mathematics and statistics background. +- _INF4490 Biologically Inspired Computing_ http://www.uio.no/studier/emner/matnat/ifi/INF4490/. An introduction to self-adapting methods also called artificial intelligence or machine learning. +- _IN-STK5000 Adaptive Methods for Data-Based Decision Making_ https://www.uio.no/studier/emner/matnat/ifi/IN-STK5000/index-eng.html. Methods for adaptive collection and processing of data based on machine learning techniques. +- _IN5400/INF5860 Machine Learning for Image Analysis_ https://www.uio.no/studier/emner/matnat/ifi/IN5400/. An introduction to deep learning with particular emphasis on applications within Image analysis, but useful for other application areas too. +- _TEK5040 Deep learning for autonomous systems_ https://www.uio.no/studier/emner/matnat/its/TEK5040/. The course addresses advanced algorithms and architectures for deep learning with neural networks. The course provides an introduction to how deep-learning techniques can be used in the construction of key parts of advanced autonomous systems that exist in physical environments and cyber environments. +- _STK4051 Computational Statistics_ https://www.uio.no/studier/emner/matnat/math/STK4051/index-eng.html +- _STK4021 Applied Bayesian Analysis and Numerical Methods_ https://www.uio.no/studier/emner/matnat/math/STK4021/ + + diff --git a/doc/LectureNotes/_build/html/_static/__init__.py b/doc/LectureNotes/_build/html/_static/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/doc/LectureNotes/_build/html/_static/__pycache__/__init__.cpython-38.pyc b/doc/LectureNotes/_build/html/_static/__pycache__/__init__.cpython-38.pyc new file mode 100644 index 0000000000000000000000000000000000000000..dce0d97cb2a0cc96a29090e51370f0c4fbbac3fb GIT binary patch literal 185 zcmYj~F%H5o5Ck2G0wM7b3UWmn3Ix0$4N8k~k`p#b&Q|V_$dmXIEl;4M!mePYoz-r$ z)pEH|QF~jSQ@#@ZmBn(1=2=9mj%t;a4>hLwhtCNr#*tyLS0qLP9|R1U##3tw=v@tA z66>kRH^5GC9Zb`i3o>x9j_$hlzSClHKwvTA8qnI26RqGvJ2zT6Qo AJOBUy literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/html/_static/basic.css b/doc/LectureNotes/_build/html/_static/basic.css new file mode 100644 index 000000000..fb51eb711 --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/basic.css @@ -0,0 +1,856 @@ +/* + * basic.css + * ~~~~~~~~~ + * + * Sphinx stylesheet -- basic theme. + * + * :copyright: Copyright 2007-2020 by the Sphinx team, see AUTHORS. + * :license: BSD, see LICENSE for details. + * + */ + +/* -- main layout ----------------------------------------------------------- */ + +div.clearer { + clear: both; +} + +div.section::after { + display: block; + content: ''; + clear: left; +} + +/* -- relbar ---------------------------------------------------------------- */ + +div.related { + width: 100%; + font-size: 90%; +} + +div.related h3 { + display: none; +} + +div.related ul { + margin: 0; + padding: 0 0 0 10px; + list-style: none; +} + +div.related li { + display: inline; +} + +div.related li.right { + float: right; + margin-right: 5px; +} + +/* -- sidebar --------------------------------------------------------------- */ + +div.sphinxsidebarwrapper { + padding: 10px 5px 0 10px; +} + +div.sphinxsidebar { + float: left; + width: 270px; + margin-left: -100%; + font-size: 90%; + word-wrap: break-word; + overflow-wrap : break-word; +} + +div.sphinxsidebar ul { + list-style: none; +} + +div.sphinxsidebar ul ul, +div.sphinxsidebar ul.want-points { + margin-left: 20px; + list-style: square; +} + +div.sphinxsidebar ul ul { + margin-top: 0; + margin-bottom: 0; +} + +div.sphinxsidebar form { + margin-top: 10px; +} + +div.sphinxsidebar input { + border: 1px solid #98dbcc; + font-family: sans-serif; + font-size: 1em; +} + +div.sphinxsidebar #searchbox form.search { + overflow: hidden; +} + +div.sphinxsidebar #searchbox input[type="text"] { + float: left; + width: 80%; + padding: 0.25em; + box-sizing: border-box; +} + +div.sphinxsidebar #searchbox input[type="submit"] { + float: left; + width: 20%; + border-left: none; + padding: 0.25em; + box-sizing: border-box; +} + + +img { + border: 0; + max-width: 100%; +} + +/* -- search page ----------------------------------------------------------- */ + +ul.search { + margin: 10px 0 0 20px; + padding: 0; +} + +ul.search li { + padding: 5px 0 5px 20px; + background-image: url(file.png); + background-repeat: no-repeat; + background-position: 0 7px; +} + +ul.search li a { + font-weight: bold; +} + +ul.search li div.context { + color: #888; + margin: 2px 0 0 30px; + text-align: left; +} + +ul.keywordmatches li.goodmatch a { + font-weight: bold; +} + +/* -- index page ------------------------------------------------------------ */ + +table.contentstable { + width: 90%; + margin-left: auto; + margin-right: auto; +} + +table.contentstable p.biglink { + line-height: 150%; +} + +a.biglink { + font-size: 1.3em; +} + +span.linkdescr { + font-style: italic; + padding-top: 5px; + font-size: 90%; +} + +/* -- general index --------------------------------------------------------- */ + +table.indextable { + width: 100%; +} + +table.indextable td { + text-align: left; + vertical-align: top; +} + +table.indextable ul { + margin-top: 0; + margin-bottom: 0; + list-style-type: none; +} + +table.indextable > tbody > tr > td > ul { + padding-left: 0em; +} + +table.indextable tr.pcap { + height: 10px; +} + +table.indextable tr.cap { + margin-top: 10px; + background-color: #f2f2f2; +} + +img.toggler { + margin-right: 3px; + margin-top: 3px; + cursor: pointer; +} + +div.modindex-jumpbox { + border-top: 1px solid #ddd; + border-bottom: 1px solid #ddd; + margin: 1em 0 1em 0; + padding: 0.4em; +} + +div.genindex-jumpbox { + border-top: 1px solid #ddd; + border-bottom: 1px solid #ddd; + margin: 1em 0 1em 0; + padding: 0.4em; +} + +/* -- domain module index --------------------------------------------------- */ + +table.modindextable td { + padding: 2px; + border-collapse: collapse; +} + +/* -- general body styles --------------------------------------------------- */ + +div.body { + min-width: 450px; + max-width: 800px; +} + +div.body p, div.body dd, div.body li, div.body blockquote { + -moz-hyphens: auto; + -ms-hyphens: auto; + -webkit-hyphens: auto; + hyphens: auto; +} + +a.headerlink { + visibility: hidden; +} + +a.brackets:before, +span.brackets > a:before{ + content: "["; +} + +a.brackets:after, +span.brackets > a:after { + content: "]"; +} + +h1:hover > a.headerlink, +h2:hover > a.headerlink, +h3:hover > a.headerlink, +h4:hover > a.headerlink, +h5:hover > a.headerlink, +h6:hover > a.headerlink, +dt:hover > a.headerlink, +caption:hover > a.headerlink, +p.caption:hover > a.headerlink, +div.code-block-caption:hover > a.headerlink { + visibility: visible; +} + +div.body p.caption { + text-align: inherit; +} + +div.body td { + text-align: left; +} + +.first { + margin-top: 0 !important; +} + +p.rubric { + margin-top: 30px; + font-weight: bold; +} + +img.align-left, .figure.align-left, object.align-left { + clear: left; + float: left; + margin-right: 1em; +} + +img.align-right, .figure.align-right, object.align-right { + clear: right; + float: right; + margin-left: 1em; +} + +img.align-center, .figure.align-center, object.align-center { + display: block; + margin-left: auto; + margin-right: auto; +} + +img.align-default, .figure.align-default { + display: block; + margin-left: auto; + margin-right: auto; +} + +.align-left { + text-align: left; +} + +.align-center { + text-align: center; +} + +.align-default { + text-align: center; +} + +.align-right { + text-align: right; +} + +/* -- sidebars -------------------------------------------------------------- */ + +div.sidebar { + margin: 0 0 0.5em 1em; + border: 1px solid #ddb; + padding: 7px; + background-color: #ffe; + width: 40%; + float: right; + clear: right; + overflow-x: auto; +} + +p.sidebar-title { + font-weight: bold; +} + +div.admonition, div.topic, blockquote { + clear: left; +} + +/* -- topics ---------------------------------------------------------------- */ + +div.topic { + border: 1px solid #ccc; + padding: 7px; + margin: 10px 0 10px 0; +} + +p.topic-title { + font-size: 1.1em; + font-weight: bold; + margin-top: 10px; +} + +/* -- admonitions ----------------------------------------------------------- */ + +div.admonition { + margin-top: 10px; + margin-bottom: 10px; + padding: 7px; +} + +div.admonition dt { + font-weight: bold; +} + +p.admonition-title { + margin: 0px 10px 5px 0px; + font-weight: bold; +} + +div.body p.centered { + text-align: center; + margin-top: 25px; +} + +/* -- content of sidebars/topics/admonitions -------------------------------- */ + +div.sidebar > :last-child, +div.topic > :last-child, +div.admonition > :last-child { + margin-bottom: 0; +} + +div.sidebar::after, +div.topic::after, +div.admonition::after, +blockquote::after { + display: block; + content: ''; + clear: both; +} + +/* -- tables ---------------------------------------------------------------- */ + +table.docutils { + margin-top: 10px; + margin-bottom: 10px; + border: 0; + border-collapse: collapse; +} + +table.align-center { + margin-left: auto; + margin-right: auto; +} + +table.align-default { + margin-left: auto; + margin-right: auto; +} + +table caption span.caption-number { + font-style: italic; +} + +table caption span.caption-text { +} + +table.docutils td, table.docutils th { + padding: 1px 8px 1px 5px; + border-top: 0; + border-left: 0; + border-right: 0; + border-bottom: 1px solid #aaa; +} + +table.footnote td, table.footnote th { + border: 0 !important; +} + +th { + text-align: left; + padding-right: 5px; +} + +table.citation { + border-left: solid 1px gray; + margin-left: 1px; +} + +table.citation td { + border-bottom: none; +} + +th > :first-child, +td > :first-child { + margin-top: 0px; +} + +th > :last-child, +td > :last-child { + margin-bottom: 0px; +} + +/* -- figures --------------------------------------------------------------- */ + +div.figure { + margin: 0.5em; + padding: 0.5em; +} + +div.figure p.caption { + padding: 0.3em; +} + +div.figure p.caption span.caption-number { + font-style: italic; +} + +div.figure p.caption span.caption-text { +} + +/* -- field list styles ----------------------------------------------------- */ + +table.field-list td, table.field-list th { + border: 0 !important; +} + +.field-list ul { + margin: 0; + padding-left: 1em; +} + +.field-list p { + margin: 0; +} + +.field-name { + -moz-hyphens: manual; + -ms-hyphens: manual; + -webkit-hyphens: manual; + hyphens: manual; +} + +/* -- hlist styles ---------------------------------------------------------- */ + +table.hlist { + margin: 1em 0; +} + +table.hlist td { + vertical-align: top; +} + + +/* -- other body styles ----------------------------------------------------- */ + +ol.arabic { + list-style: decimal; +} + +ol.loweralpha { + list-style: lower-alpha; +} + +ol.upperalpha { + list-style: upper-alpha; +} + +ol.lowerroman { + list-style: lower-roman; +} + +ol.upperroman { + list-style: upper-roman; +} + +:not(li) > ol > li:first-child > :first-child, +:not(li) > ul > li:first-child > :first-child { + margin-top: 0px; +} + +:not(li) > ol > li:last-child > :last-child, +:not(li) > ul > li:last-child > :last-child { + margin-bottom: 0px; +} + +ol.simple ol p, +ol.simple ul p, +ul.simple ol p, +ul.simple ul p { + margin-top: 0; +} + +ol.simple > li:not(:first-child) > p, +ul.simple > li:not(:first-child) > p { + margin-top: 0; +} + +ol.simple p, +ul.simple p { + margin-bottom: 0; +} + +dl.footnote > dt, +dl.citation > dt { + float: left; + margin-right: 0.5em; +} + +dl.footnote > dd, +dl.citation > dd { + margin-bottom: 0em; +} + +dl.footnote > dd:after, +dl.citation > dd:after { + content: ""; + clear: both; +} + +dl.field-list { + display: grid; + grid-template-columns: fit-content(30%) auto; +} + +dl.field-list > dt { + font-weight: bold; + word-break: break-word; + padding-left: 0.5em; + padding-right: 5px; +} + +dl.field-list > dt:after { + content: ":"; +} + +dl.field-list > dd { + padding-left: 0.5em; + margin-top: 0em; + margin-left: 0em; + margin-bottom: 0em; +} + +dl { + margin-bottom: 15px; +} + +dd > :first-child { + margin-top: 0px; +} + +dd ul, dd table { + margin-bottom: 10px; +} + +dd { + margin-top: 3px; + margin-bottom: 10px; + margin-left: 30px; +} + +dl > dd:last-child, +dl > dd:last-child > :last-child { + margin-bottom: 0; +} + +dt:target, span.highlighted { + background-color: #fbe54e; +} + +rect.highlighted { + fill: #fbe54e; +} + +dl.glossary dt { + font-weight: bold; + font-size: 1.1em; +} + +.optional { + font-size: 1.3em; +} + +.sig-paren { + font-size: larger; +} + +.versionmodified { + font-style: italic; +} + +.system-message { + background-color: #fda; + padding: 5px; + border: 3px solid red; +} + +.footnote:target { + background-color: #ffa; +} + +.line-block { + display: block; + margin-top: 1em; + margin-bottom: 1em; +} + +.line-block .line-block { + margin-top: 0; + margin-bottom: 0; + margin-left: 1.5em; +} + +.guilabel, .menuselection { + font-family: sans-serif; +} + +.accelerator { + text-decoration: underline; +} + +.classifier { + font-style: oblique; +} + +.classifier:before { + font-style: normal; + margin: 0.5em; + content: ":"; +} + +abbr, acronym { + border-bottom: dotted 1px; + cursor: help; +} + +/* -- code displays --------------------------------------------------------- */ + +pre { + overflow: auto; + overflow-y: hidden; /* fixes display issues on Chrome browsers */ +} + +pre, div[class*="highlight-"] { + clear: both; +} + +span.pre { + -moz-hyphens: none; + -ms-hyphens: none; + -webkit-hyphens: none; + hyphens: none; +} + +div[class*="highlight-"] { + margin: 1em 0; +} + +td.linenos pre { + border: 0; + background-color: transparent; + color: #aaa; +} + +table.highlighttable { + display: block; +} + +table.highlighttable tbody { + display: block; +} + +table.highlighttable tr { + display: flex; +} + +table.highlighttable td { + margin: 0; + padding: 0; +} + +table.highlighttable td.linenos { + padding-right: 0.5em; +} + +table.highlighttable td.code { + flex: 1; + overflow: hidden; +} + +.highlight .hll { + display: block; +} + +div.highlight pre, +table.highlighttable pre { + margin: 0; +} + +div.code-block-caption + div { + margin-top: 0; +} + +div.code-block-caption { + margin-top: 1em; + padding: 2px 5px; + font-size: small; +} + +div.code-block-caption code { + background-color: transparent; +} + +table.highlighttable td.linenos, +span.linenos, +div.doctest > div.highlight span.gp { /* gp: Generic.Prompt */ + user-select: none; +} + +div.code-block-caption span.caption-number { + padding: 0.1em 0.3em; + font-style: italic; +} + +div.code-block-caption span.caption-text { +} + +div.literal-block-wrapper { + margin: 1em 0; +} + +code.descname { + background-color: transparent; + font-weight: bold; + font-size: 1.2em; +} + +code.descclassname { + background-color: transparent; +} + +code.xref, a code { + background-color: transparent; + font-weight: bold; +} + +h1 code, h2 code, h3 code, h4 code, h5 code, h6 code { + background-color: transparent; +} + +.viewcode-link { + float: right; +} + +.viewcode-back { + float: right; + font-family: sans-serif; +} + +div.viewcode-block:target { + margin: -1px -10px; + padding: 0 10px; +} + +/* -- math display ---------------------------------------------------------- */ + +img.math { + vertical-align: middle; +} + +div.body div.math p { + text-align: center; +} + +span.eqno { + float: right; +} + +span.eqno a.headerlink { + position: absolute; + z-index: 1; +} + +div.math:hover a.headerlink { + visibility: visible; +} + +/* -- printout stylesheet --------------------------------------------------- */ + +@media print { + div.document, + div.documentwrapper, + div.bodywrapper { + margin: 0 !important; + width: 100%; + } + + div.sphinxsidebar, + div.related, + div.footer, + #top-link { + display: none; + } +} \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/_static/clipboard.min.js b/doc/LectureNotes/_build/html/_static/clipboard.min.js new file mode 100644 index 000000000..02c549e35 --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/clipboard.min.js @@ -0,0 +1,7 @@ +/*! + * clipboard.js v2.0.4 + * https://zenorocha.github.io/clipboard.js + * + * Licensed MIT © Zeno Rocha + */ +!function(t,e){"object"==typeof exports&&"object"==typeof module?module.exports=e():"function"==typeof define&&define.amd?define([],e):"object"==typeof exports?exports.ClipboardJS=e():t.ClipboardJS=e()}(this,function(){return function(n){var o={};function r(t){if(o[t])return o[t].exports;var e=o[t]={i:t,l:!1,exports:{}};return n[t].call(e.exports,e,e.exports,r),e.l=!0,e.exports}return r.m=n,r.c=o,r.d=function(t,e,n){r.o(t,e)||Object.defineProperty(t,e,{enumerable:!0,get:n})},r.r=function(t){"undefined"!=typeof Symbol&&Symbol.toStringTag&&Object.defineProperty(t,Symbol.toStringTag,{value:"Module"}),Object.defineProperty(t,"__esModule",{value:!0})},r.t=function(e,t){if(1&t&&(e=r(e)),8&t)return e;if(4&t&&"object"==typeof e&&e&&e.__esModule)return e;var n=Object.create(null);if(r.r(n),Object.defineProperty(n,"default",{enumerable:!0,value:e}),2&t&&"string"!=typeof e)for(var o in e)r.d(n,o,function(t){return e[t]}.bind(null,o));return n},r.n=function(t){var e=t&&t.__esModule?function(){return t.default}:function(){return t};return r.d(e,"a",e),e},r.o=function(t,e){return Object.prototype.hasOwnProperty.call(t,e)},r.p="",r(r.s=0)}([function(t,e,n){"use strict";var r="function"==typeof Symbol&&"symbol"==typeof Symbol.iterator?function(t){return typeof t}:function(t){return t&&"function"==typeof Symbol&&t.constructor===Symbol&&t!==Symbol.prototype?"symbol":typeof t},i=function(){function o(t,e){for(var n=0;n + + + + diff --git a/doc/LectureNotes/_build/html/_static/copybutton.css b/doc/LectureNotes/_build/html/_static/copybutton.css new file mode 100644 index 000000000..75b17a83d --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/copybutton.css @@ -0,0 +1,67 @@ +/* Copy buttons */ +a.copybtn { + position: absolute; + top: .2em; + right: .2em; + width: 1em; + height: 1em; + opacity: .3; + transition: opacity 0.5s; + border: none; + user-select: none; +} + +div.highlight { + position: relative; +} + +a.copybtn > img { + vertical-align: top; + margin: 0; + top: 0; + left: 0; + position: absolute; +} + +.highlight:hover .copybtn { + opacity: 1; +} + +/** + * A minimal CSS-only tooltip copied from: + * https://codepen.io/mildrenben/pen/rVBrpK + * + * To use, write HTML like the following: + * + *

Short

+ */ + .o-tooltip--left { + position: relative; + } + + .o-tooltip--left:after { + opacity: 0; + visibility: hidden; + position: absolute; + content: attr(data-tooltip); + padding: 2px; + top: 0; + left: -.2em; + background: grey; + font-size: 1rem; + color: white; + white-space: nowrap; + z-index: 2; + border-radius: 2px; + transform: translateX(-102%) translateY(0); + transition: opacity 0.2s cubic-bezier(0.64, 0.09, 0.08, 1), transform 0.2s cubic-bezier(0.64, 0.09, 0.08, 1); +} + +.o-tooltip--left:hover:after { + display: block; + opacity: 1; + visibility: visible; + transform: translateX(-100%) translateY(0); + transition: opacity 0.2s cubic-bezier(0.64, 0.09, 0.08, 1), transform 0.2s cubic-bezier(0.64, 0.09, 0.08, 1); + transition-delay: .5s; +} diff --git a/doc/LectureNotes/_build/html/_static/copybutton.js b/doc/LectureNotes/_build/html/_static/copybutton.js new file mode 100644 index 000000000..65a59167a --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/copybutton.js @@ -0,0 +1,153 @@ +// Localization support +const messages = { + 'en': { + 'copy': 'Copy', + 'copy_to_clipboard': 'Copy to clipboard', + 'copy_success': 'Copied!', + 'copy_failure': 'Failed to copy', + }, + 'es' : { + 'copy': 'Copiar', + 'copy_to_clipboard': 'Copiar al portapapeles', + 'copy_success': '¡Copiado!', + 'copy_failure': 'Error al copiar', + }, + 'de' : { + 'copy': 'Kopieren', + 'copy_to_clipboard': 'In die Zwischenablage kopieren', + 'copy_success': 'Kopiert!', + 'copy_failure': 'Fehler beim Kopieren', + } +} + +let locale = 'en' +if( document.documentElement.lang !== undefined + && messages[document.documentElement.lang] !== undefined ) { + locale = document.documentElement.lang +} + +/** + * Set up copy/paste for code blocks + */ + +const runWhenDOMLoaded = cb => { + if (document.readyState != 'loading') { + cb() + } else if (document.addEventListener) { + document.addEventListener('DOMContentLoaded', cb) + } else { + document.attachEvent('onreadystatechange', function() { + if (document.readyState == 'complete') cb() + }) + } +} + +const codeCellId = index => `codecell${index}` + +// Clears selected text since ClipboardJS will select the text when copying +const clearSelection = () => { + if (window.getSelection) { + window.getSelection().removeAllRanges() + } else if (document.selection) { + document.selection.empty() + } +} + +// Changes tooltip text for two seconds, then changes it back +const temporarilyChangeTooltip = (el, newText) => { + const oldText = el.getAttribute('data-tooltip') + el.setAttribute('data-tooltip', newText) + setTimeout(() => el.setAttribute('data-tooltip', oldText), 2000) +} + +const addCopyButtonToCodeCells = () => { + // If ClipboardJS hasn't loaded, wait a bit and try again. This + // happens because we load ClipboardJS asynchronously. + if (window.ClipboardJS === undefined) { + setTimeout(addCopyButtonToCodeCells, 250) + return + } + + // Add copybuttons to all of our code cells + const codeCells = document.querySelectorAll('div.highlight pre') + codeCells.forEach((codeCell, index) => { + const id = codeCellId(index) + codeCell.setAttribute('id', id) + const pre_bg = getComputedStyle(codeCell).backgroundColor; + + const clipboardButton = id => + `
+ ${messages[locale]['copy_to_clipboard']} + ` + codeCell.insertAdjacentHTML('afterend', clipboardButton(id)) + }) + +function escapeRegExp(string) { + return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); // $& means the whole matched string +} + +// Callback when a copy button is clicked. Will be passed the node that was clicked +// should then grab the text and replace pieces of text that shouldn't be used in output +function formatCopyText(textContent, copybuttonPromptText, isRegexp = false, onlyCopyPromptLines = true, removePrompts = true) { + + var regexp; + var match; + + // create regexp to capture prompt and remaining line + if (isRegexp) { + regexp = new RegExp('^(' + copybuttonPromptText + ')(.*)') + } else { + regexp = new RegExp('^(' + escapeRegExp(copybuttonPromptText) + ')(.*)') + } + + const outputLines = []; + var promptFound = false; + for (const line of textContent.split('\n')) { + match = line.match(regexp) + if (match) { + promptFound = true + if (removePrompts) { + outputLines.push(match[2]) + } else { + outputLines.push(line) + } + } else { + if (!onlyCopyPromptLines) { + outputLines.push(line) + } + } + } + + // If no lines with the prompt were found then just use original lines + if (promptFound) { + textContent = outputLines.join('\n'); + } + + // Remove a trailing newline to avoid auto-running when pasting + if (textContent.endsWith("\n")) { + textContent = textContent.slice(0, -1) + } + return textContent +} + + +var copyTargetText = (trigger) => { + var target = document.querySelector(trigger.attributes['data-clipboard-target'].value); + return formatCopyText(target.innerText, '', false, true, true) +} + + // Initialize with a callback so we can modify the text before copy + const clipboard = new ClipboardJS('.copybtn', {text: copyTargetText}) + + // Update UI with error/success messages + clipboard.on('success', event => { + clearSelection() + temporarilyChangeTooltip(event.trigger, messages[locale]['copy_success']) + }) + + clipboard.on('error', event => { + temporarilyChangeTooltip(event.trigger, messages[locale]['copy_failure']) + }) +} + +runWhenDOMLoaded(addCopyButtonToCodeCells) \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/_static/copybutton_funcs.js b/doc/LectureNotes/_build/html/_static/copybutton_funcs.js new file mode 100644 index 000000000..57caa5585 --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/copybutton_funcs.js @@ -0,0 +1,47 @@ +function escapeRegExp(string) { + return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); // $& means the whole matched string +} + +// Callback when a copy button is clicked. Will be passed the node that was clicked +// should then grab the text and replace pieces of text that shouldn't be used in output +export function formatCopyText(textContent, copybuttonPromptText, isRegexp = false, onlyCopyPromptLines = true, removePrompts = true) { + + var regexp; + var match; + + // create regexp to capture prompt and remaining line + if (isRegexp) { + regexp = new RegExp('^(' + copybuttonPromptText + ')(.*)') + } else { + regexp = new RegExp('^(' + escapeRegExp(copybuttonPromptText) + ')(.*)') + } + + const outputLines = []; + var promptFound = false; + for (const line of textContent.split('\n')) { + match = line.match(regexp) + if (match) { + promptFound = true + if (removePrompts) { + outputLines.push(match[2]) + } else { + outputLines.push(line) + } + } else { + if (!onlyCopyPromptLines) { + outputLines.push(line) + } + } + } + + // If no lines with the prompt were found then just use original lines + if (promptFound) { + textContent = outputLines.join('\n'); + } + + // Remove a trailing newline to avoid auto-running when pasting + if (textContent.endsWith("\n")) { + textContent = textContent.slice(0, -1) + } + return textContent +} diff --git a/doc/LectureNotes/_build/html/_static/css/index.f658d18f9b420779cfdf24aa0a7e2d77.css b/doc/LectureNotes/_build/html/_static/css/index.f658d18f9b420779cfdf24aa0a7e2d77.css new file mode 100644 index 000000000..7fd19a770 --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/css/index.f658d18f9b420779cfdf24aa0a7e2d77.css @@ -0,0 +1,6 @@ +/*! + * Bootstrap v4.5.0 (https://getbootstrap.com/) + * Copyright 2011-2020 The Bootstrap Authors + * Copyright 2011-2020 Twitter, Inc. + * Licensed under MIT (https://github.com/twbs/bootstrap/blob/master/LICENSE) + */:root{--blue:#007bff;--indigo:#6610f2;--purple:#6f42c1;--pink:#e83e8c;--red:#dc3545;--orange:#fd7e14;--yellow:#ffc107;--green:#28a745;--teal:#20c997;--cyan:#17a2b8;--white:#fff;--gray:#6c757d;--gray-dark:#343a40;--primary:#007bff;--secondary:#6c757d;--success:#28a745;--info:#17a2b8;--warning:#ffc107;--danger:#dc3545;--light:#f8f9fa;--dark:#343a40;--breakpoint-xs:0;--breakpoint-sm:576px;--breakpoint-md:768px;--breakpoint-lg:992px;--breakpoint-xl:1200px;--font-family-sans-serif:-apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,"Helvetica Neue",Arial,"Noto Sans",sans-serif,"Apple Color Emoji","Segoe UI Emoji","Segoe UI Symbol","Noto Color Emoji";--font-family-monospace:SFMono-Regular,Menlo,Monaco,Consolas,"Liberation Mono","Courier New",monospace}*,:after,:before{box-sizing:border-box}html{font-family:sans-serif;line-height:1.15;-webkit-text-size-adjust:100%;-webkit-tap-highlight-color:rgba(0,0,0,0)}article,aside,figcaption,figure,footer,header,hgroup,main,nav,section{display:block}body{margin:0;font-family:-apple-system,BlinkMacSystemFont,Segoe UI,Roboto,Helvetica Neue,Arial,Noto Sans,sans-serif,Apple Color Emoji,Segoe UI Emoji,Segoe UI Symbol,Noto Color Emoji;font-size:1rem;line-height:1.5;color:#212529;text-align:left}[tabindex="-1"]:focus:not(:focus-visible){outline:0!important}hr{box-sizing:content-box;height:0;overflow:visible}h1,h2,h3,h4,h5,h6{margin-top:0;margin-bottom:.5rem}p{margin-top:0;margin-bottom:1rem}abbr[data-original-title],abbr[title]{text-decoration:underline;text-decoration:underline dotted;cursor:help;border-bottom:0;text-decoration-skip-ink:none}address{font-style:normal;line-height:inherit}address,dl,ol,ul{margin-bottom:1rem}dl,ol,ul{margin-top:0}ol ol,ol ul,ul ol,ul ul{margin-bottom:0}dt{font-weight:700}dd{margin-bottom:.5rem;margin-left:0}blockquote{margin:0 0 1rem}b,strong{font-weight:bolder}small{font-size:80%}sub,sup{position:relative;font-size:75%;line-height:0;vertical-align:baseline}sub{bottom:-.25em}sup{top:-.5em}a{color:#007bff;background-color:transparent}a:hover{color:#0056b3}a:not([href]),a:not([href]):hover{color:inherit;text-decoration:none}code,kbd,pre,samp{font-family:SFMono-Regular,Menlo,Monaco,Consolas,Liberation Mono,Courier New,monospace;font-size:1em}pre{margin-top:0;margin-bottom:1rem;overflow:auto;-ms-overflow-style:scrollbar}figure{margin:0 0 1rem}img{border-style:none}img,svg{vertical-align:middle}svg{overflow:hidden}table{border-collapse:collapse}caption{padding-top:.75rem;padding-bottom:.75rem;color:#6c757d;text-align:left;caption-side:bottom}th{text-align:inherit}label{display:inline-block;margin-bottom:.5rem}button{border-radius:0}button:focus{outline:1px dotted;outline:5px auto -webkit-focus-ring-color}button,input,optgroup,select,textarea{margin:0;font-family:inherit;font-size:inherit;line-height:inherit}button,input{overflow:visible}button,select{text-transform:none}[role=button]{cursor:pointer}select{word-wrap:normal}[type=button],[type=reset],[type=submit],button{-webkit-appearance:button}[type=button]:not(:disabled),[type=reset]:not(:disabled),[type=submit]:not(:disabled),button:not(:disabled){cursor:pointer}[type=button]::-moz-focus-inner,[type=reset]::-moz-focus-inner,[type=submit]::-moz-focus-inner,button::-moz-focus-inner{padding:0;border-style:none}input[type=checkbox],input[type=radio]{box-sizing:border-box;padding:0}textarea{overflow:auto;resize:vertical}fieldset{min-width:0;padding:0;margin:0;border:0}legend{display:block;width:100%;max-width:100%;padding:0;margin-bottom:.5rem;font-size:1.5rem;line-height:inherit;color:inherit;white-space:normal}progress{vertical-align:baseline}[type=number]::-webkit-inner-spin-button,[type=number]::-webkit-outer-spin-button{height:auto}[type=search]{outline-offset:-2px;-webkit-appearance:none}[type=search]::-webkit-search-decoration{-webkit-appearance:none}::-webkit-file-upload-button{font:inherit;-webkit-appearance:button}output{display:inline-block}summary{display:list-item;cursor:pointer}template{display:none}[hidden]{display:none!important}.h1,.h2,.h3,.h4,.h5,.h6,h1,h2,h3,h4,h5,h6{margin-bottom:.5rem;font-weight:500;line-height:1.2}.h1,h1{font-size:2.5rem}.h2,h2{font-size:2rem}.h3,h3{font-size:1.75rem}.h4,h4{font-size:1.5rem}.h5,h5{font-size:1.25rem}.h6,h6{font-size:1rem}.lead{font-size:1.25rem;font-weight:300}.display-1{font-size:6rem}.display-1,.display-2{font-weight:300;line-height:1.2}.display-2{font-size:5.5rem}.display-3{font-size:4.5rem}.display-3,.display-4{font-weight:300;line-height:1.2}.display-4{font-size:3.5rem}hr{margin-top:1rem;margin-bottom:1rem;border-top:1px solid rgba(0,0,0,.1)}.small,small{font-size:80%;font-weight:400}.mark,mark{padding:.2em;background-color:#fcf8e3}.list-inline,.list-unstyled{padding-left:0;list-style:none}.list-inline-item{display:inline-block}.list-inline-item:not(:last-child){margin-right:.5rem}.initialism{font-size:90%;text-transform:uppercase}.blockquote{margin-bottom:1rem;font-size:1.25rem}.blockquote-footer{display:block;font-size:80%;color:#6c757d}.blockquote-footer:before{content:"\2014\00A0"}.img-fluid,.img-thumbnail{max-width:100%;height:auto}.img-thumbnail{padding:.25rem;background-color:#fff;border:1px solid #dee2e6;border-radius:.25rem}.figure{display:inline-block}.figure-img{margin-bottom:.5rem;line-height:1}.figure-caption{font-size:90%;color:#6c757d}code{font-size:87.5%;color:#e83e8c;word-wrap:break-word}a>code{color:inherit}kbd{padding:.2rem .4rem;font-size:87.5%;color:#fff;background-color:#212529;border-radius:.2rem}kbd kbd{padding:0;font-size:100%;font-weight:700}pre{display:block;font-size:87.5%;color:#212529}pre code{font-size:inherit;color:inherit;word-break:normal}.pre-scrollable{max-height:340px;overflow-y:scroll}.container{width:100%;padding-right:15px;padding-left:15px;margin-right:auto;margin-left:auto}@media (min-width:576px){.container{max-width:540px}}@media (min-width:768px){.container{max-width:720px}}@media (min-width:992px){.container{max-width:960px}}@media (min-width:1200px){.container{max-width:1400px}}.container-fluid,.container-lg,.container-md,.container-sm,.container-xl{width:100%;padding-right:15px;padding-left:15px;margin-right:auto;margin-left:auto}@media (min-width:576px){.container,.container-sm{max-width:540px}}@media (min-width:768px){.container,.container-md,.container-sm{max-width:720px}}@media (min-width:992px){.container,.container-lg,.container-md,.container-sm{max-width:960px}}@media (min-width:1200px){.container,.container-lg,.container-md,.container-sm,.container-xl{max-width:1400px}}.row{display:flex;flex-wrap:wrap;margin-right:-15px;margin-left:-15px}.no-gutters{margin-right:0;margin-left:0}.no-gutters>.col,.no-gutters>[class*=col-]{padding-right:0;padding-left:0}.col,.col-1,.col-2,.col-3,.col-4,.col-5,.col-6,.col-7,.col-8,.col-9,.col-10,.col-11,.col-12,.col-auto,.col-lg,.col-lg-1,.col-lg-2,.col-lg-3,.col-lg-4,.col-lg-5,.col-lg-6,.col-lg-7,.col-lg-8,.col-lg-9,.col-lg-10,.col-lg-11,.col-lg-12,.col-lg-auto,.col-md,.col-md-1,.col-md-2,.col-md-3,.col-md-4,.col-md-5,.col-md-6,.col-md-7,.col-md-8,.col-md-9,.col-md-10,.col-md-11,.col-md-12,.col-md-auto,.col-sm,.col-sm-1,.col-sm-2,.col-sm-3,.col-sm-4,.col-sm-5,.col-sm-6,.col-sm-7,.col-sm-8,.col-sm-9,.col-sm-10,.col-sm-11,.col-sm-12,.col-sm-auto,.col-xl,.col-xl-1,.col-xl-2,.col-xl-3,.col-xl-4,.col-xl-5,.col-xl-6,.col-xl-7,.col-xl-8,.col-xl-9,.col-xl-10,.col-xl-11,.col-xl-12,.col-xl-auto{position:relative;width:100%;padding-right:15px;padding-left:15px}.col{flex-basis:0;flex-grow:1;min-width:0;max-width:100%}.row-cols-1>*{flex:0 0 100%;max-width:100%}.row-cols-2>*{flex:0 0 50%;max-width:50%}.row-cols-3>*{flex:0 0 33.33333%;max-width:33.33333%}.row-cols-4>*{flex:0 0 25%;max-width:25%}.row-cols-5>*{flex:0 0 20%;max-width:20%}.row-cols-6>*{flex:0 0 16.66667%;max-width:16.66667%}.col-auto{flex:0 0 auto;width:auto;max-width:100%}.col-1{flex:0 0 8.33333%;max-width:8.33333%}.col-2{flex:0 0 16.66667%;max-width:16.66667%}.col-3{flex:0 0 25%;max-width:25%}.col-4{flex:0 0 33.33333%;max-width:33.33333%}.col-5{flex:0 0 41.66667%;max-width:41.66667%}.col-6{flex:0 0 50%;max-width:50%}.col-7{flex:0 0 58.33333%;max-width:58.33333%}.col-8{flex:0 0 66.66667%;max-width:66.66667%}.col-9{flex:0 0 75%;max-width:75%}.col-10{flex:0 0 83.33333%;max-width:83.33333%}.col-11{flex:0 0 91.66667%;max-width:91.66667%}.col-12{flex:0 0 100%;max-width:100%}.order-first{order:-1}.order-last{order:13}.order-0{order:0}.order-1{order:1}.order-2{order:2}.order-3{order:3}.order-4{order:4}.order-5{order:5}.order-6{order:6}.order-7{order:7}.order-8{order:8}.order-9{order:9}.order-10{order:10}.order-11{order:11}.order-12{order:12}.offset-1{margin-left:8.33333%}.offset-2{margin-left:16.66667%}.offset-3{margin-left:25%}.offset-4{margin-left:33.33333%}.offset-5{margin-left:41.66667%}.offset-6{margin-left:50%}.offset-7{margin-left:58.33333%}.offset-8{margin-left:66.66667%}.offset-9{margin-left:75%}.offset-10{margin-left:83.33333%}.offset-11{margin-left:91.66667%}@media (min-width:576px){.col-sm{flex-basis:0;flex-grow:1;min-width:0;max-width:100%}.row-cols-sm-1>*{flex:0 0 100%;max-width:100%}.row-cols-sm-2>*{flex:0 0 50%;max-width:50%}.row-cols-sm-3>*{flex:0 0 33.33333%;max-width:33.33333%}.row-cols-sm-4>*{flex:0 0 25%;max-width:25%}.row-cols-sm-5>*{flex:0 0 20%;max-width:20%}.row-cols-sm-6>*{flex:0 0 16.66667%;max-width:16.66667%}.col-sm-auto{flex:0 0 auto;width:auto;max-width:100%}.col-sm-1{flex:0 0 8.33333%;max-width:8.33333%}.col-sm-2{flex:0 0 16.66667%;max-width:16.66667%}.col-sm-3{flex:0 0 25%;max-width:25%}.col-sm-4{flex:0 0 33.33333%;max-width:33.33333%}.col-sm-5{flex:0 0 41.66667%;max-width:41.66667%}.col-sm-6{flex:0 0 50%;max-width:50%}.col-sm-7{flex:0 0 58.33333%;max-width:58.33333%}.col-sm-8{flex:0 0 66.66667%;max-width:66.66667%}.col-sm-9{flex:0 0 75%;max-width:75%}.col-sm-10{flex:0 0 83.33333%;max-width:83.33333%}.col-sm-11{flex:0 0 91.66667%;max-width:91.66667%}.col-sm-12{flex:0 0 100%;max-width:100%}.order-sm-first{order:-1}.order-sm-last{order:13}.order-sm-0{order:0}.order-sm-1{order:1}.order-sm-2{order:2}.order-sm-3{order:3}.order-sm-4{order:4}.order-sm-5{order:5}.order-sm-6{order:6}.order-sm-7{order:7}.order-sm-8{order:8}.order-sm-9{order:9}.order-sm-10{order:10}.order-sm-11{order:11}.order-sm-12{order:12}.offset-sm-0{margin-left:0}.offset-sm-1{margin-left:8.33333%}.offset-sm-2{margin-left:16.66667%}.offset-sm-3{margin-left:25%}.offset-sm-4{margin-left:33.33333%}.offset-sm-5{margin-left:41.66667%}.offset-sm-6{margin-left:50%}.offset-sm-7{margin-left:58.33333%}.offset-sm-8{margin-left:66.66667%}.offset-sm-9{margin-left:75%}.offset-sm-10{margin-left:83.33333%}.offset-sm-11{margin-left:91.66667%}}@media (min-width:768px){.col-md{flex-basis:0;flex-grow:1;min-width:0;max-width:100%}.row-cols-md-1>*{flex:0 0 100%;max-width:100%}.row-cols-md-2>*{flex:0 0 50%;max-width:50%}.row-cols-md-3>*{flex:0 0 33.33333%;max-width:33.33333%}.row-cols-md-4>*{flex:0 0 25%;max-width:25%}.row-cols-md-5>*{flex:0 0 20%;max-width:20%}.row-cols-md-6>*{flex:0 0 16.66667%;max-width:16.66667%}.col-md-auto{flex:0 0 auto;width:auto;max-width:100%}.col-md-1{flex:0 0 8.33333%;max-width:8.33333%}.col-md-2{flex:0 0 16.66667%;max-width:16.66667%}.col-md-3{flex:0 0 25%;max-width:25%}.col-md-4{flex:0 0 33.33333%;max-width:33.33333%}.col-md-5{flex:0 0 41.66667%;max-width:41.66667%}.col-md-6{flex:0 0 50%;max-width:50%}.col-md-7{flex:0 0 58.33333%;max-width:58.33333%}.col-md-8{flex:0 0 66.66667%;max-width:66.66667%}.col-md-9{flex:0 0 75%;max-width:75%}.col-md-10{flex:0 0 83.33333%;max-width:83.33333%}.col-md-11{flex:0 0 91.66667%;max-width:91.66667%}.col-md-12{flex:0 0 100%;max-width:100%}.order-md-first{order:-1}.order-md-last{order:13}.order-md-0{order:0}.order-md-1{order:1}.order-md-2{order:2}.order-md-3{order:3}.order-md-4{order:4}.order-md-5{order:5}.order-md-6{order:6}.order-md-7{order:7}.order-md-8{order:8}.order-md-9{order:9}.order-md-10{order:10}.order-md-11{order:11}.order-md-12{order:12}.offset-md-0{margin-left:0}.offset-md-1{margin-left:8.33333%}.offset-md-2{margin-left:16.66667%}.offset-md-3{margin-left:25%}.offset-md-4{margin-left:33.33333%}.offset-md-5{margin-left:41.66667%}.offset-md-6{margin-left:50%}.offset-md-7{margin-left:58.33333%}.offset-md-8{margin-left:66.66667%}.offset-md-9{margin-left:75%}.offset-md-10{margin-left:83.33333%}.offset-md-11{margin-left:91.66667%}}@media (min-width:992px){.col-lg{flex-basis:0;flex-grow:1;min-width:0;max-width:100%}.row-cols-lg-1>*{flex:0 0 100%;max-width:100%}.row-cols-lg-2>*{flex:0 0 50%;max-width:50%}.row-cols-lg-3>*{flex:0 0 33.33333%;max-width:33.33333%}.row-cols-lg-4>*{flex:0 0 25%;max-width:25%}.row-cols-lg-5>*{flex:0 0 20%;max-width:20%}.row-cols-lg-6>*{flex:0 0 16.66667%;max-width:16.66667%}.col-lg-auto{flex:0 0 auto;width:auto;max-width:100%}.col-lg-1{flex:0 0 8.33333%;max-width:8.33333%}.col-lg-2{flex:0 0 16.66667%;max-width:16.66667%}.col-lg-3{flex:0 0 25%;max-width:25%}.col-lg-4{flex:0 0 33.33333%;max-width:33.33333%}.col-lg-5{flex:0 0 41.66667%;max-width:41.66667%}.col-lg-6{flex:0 0 50%;max-width:50%}.col-lg-7{flex:0 0 58.33333%;max-width:58.33333%}.col-lg-8{flex:0 0 66.66667%;max-width:66.66667%}.col-lg-9{flex:0 0 75%;max-width:75%}.col-lg-10{flex:0 0 83.33333%;max-width:83.33333%}.col-lg-11{flex:0 0 91.66667%;max-width:91.66667%}.col-lg-12{flex:0 0 100%;max-width:100%}.order-lg-first{order:-1}.order-lg-last{order:13}.order-lg-0{order:0}.order-lg-1{order:1}.order-lg-2{order:2}.order-lg-3{order:3}.order-lg-4{order:4}.order-lg-5{order:5}.order-lg-6{order:6}.order-lg-7{order:7}.order-lg-8{order:8}.order-lg-9{order:9}.order-lg-10{order:10}.order-lg-11{order:11}.order-lg-12{order:12}.offset-lg-0{margin-left:0}.offset-lg-1{margin-left:8.33333%}.offset-lg-2{margin-left:16.66667%}.offset-lg-3{margin-left:25%}.offset-lg-4{margin-left:33.33333%}.offset-lg-5{margin-left:41.66667%}.offset-lg-6{margin-left:50%}.offset-lg-7{margin-left:58.33333%}.offset-lg-8{margin-left:66.66667%}.offset-lg-9{margin-left:75%}.offset-lg-10{margin-left:83.33333%}.offset-lg-11{margin-left:91.66667%}}@media (min-width:1200px){.col-xl{flex-basis:0;flex-grow:1;min-width:0;max-width:100%}.row-cols-xl-1>*{flex:0 0 100%;max-width:100%}.row-cols-xl-2>*{flex:0 0 50%;max-width:50%}.row-cols-xl-3>*{flex:0 0 33.33333%;max-width:33.33333%}.row-cols-xl-4>*{flex:0 0 25%;max-width:25%}.row-cols-xl-5>*{flex:0 0 20%;max-width:20%}.row-cols-xl-6>*{flex:0 0 16.66667%;max-width:16.66667%}.col-xl-auto{flex:0 0 auto;width:auto;max-width:100%}.col-xl-1{flex:0 0 8.33333%;max-width:8.33333%}.col-xl-2{flex:0 0 16.66667%;max-width:16.66667%}.col-xl-3{flex:0 0 25%;max-width:25%}.col-xl-4{flex:0 0 33.33333%;max-width:33.33333%}.col-xl-5{flex:0 0 41.66667%;max-width:41.66667%}.col-xl-6{flex:0 0 50%;max-width:50%}.col-xl-7{flex:0 0 58.33333%;max-width:58.33333%}.col-xl-8{flex:0 0 66.66667%;max-width:66.66667%}.col-xl-9{flex:0 0 75%;max-width:75%}.col-xl-10{flex:0 0 83.33333%;max-width:83.33333%}.col-xl-11{flex:0 0 91.66667%;max-width:91.66667%}.col-xl-12{flex:0 0 100%;max-width:100%}.order-xl-first{order:-1}.order-xl-last{order:13}.order-xl-0{order:0}.order-xl-1{order:1}.order-xl-2{order:2}.order-xl-3{order:3}.order-xl-4{order:4}.order-xl-5{order:5}.order-xl-6{order:6}.order-xl-7{order:7}.order-xl-8{order:8}.order-xl-9{order:9}.order-xl-10{order:10}.order-xl-11{order:11}.order-xl-12{order:12}.offset-xl-0{margin-left:0}.offset-xl-1{margin-left:8.33333%}.offset-xl-2{margin-left:16.66667%}.offset-xl-3{margin-left:25%}.offset-xl-4{margin-left:33.33333%}.offset-xl-5{margin-left:41.66667%}.offset-xl-6{margin-left:50%}.offset-xl-7{margin-left:58.33333%}.offset-xl-8{margin-left:66.66667%}.offset-xl-9{margin-left:75%}.offset-xl-10{margin-left:83.33333%}.offset-xl-11{margin-left:91.66667%}}.table{width:100%;margin-bottom:1rem;color:#212529}.table td,.table th{padding:.75rem;vertical-align:top;border-top:1px solid #dee2e6}.table thead th{vertical-align:bottom;border-bottom:2px solid #dee2e6}.table tbody+tbody{border-top:2px solid #dee2e6}.table-sm td,.table-sm th{padding:.3rem}.table-bordered,.table-bordered td,.table-bordered th{border:1px solid #dee2e6}.table-bordered thead td,.table-bordered thead th{border-bottom-width:2px}.table-borderless tbody+tbody,.table-borderless td,.table-borderless th,.table-borderless thead th{border:0}.table-striped tbody tr:nth-of-type(odd){background-color:rgba(0,0,0,.05)}.table-hover tbody tr:hover{color:#212529;background-color:rgba(0,0,0,.075)}.table-primary,.table-primary>td,.table-primary>th{background-color:#b8daff}.table-primary tbody+tbody,.table-primary td,.table-primary th,.table-primary thead th{border-color:#7abaff}.table-hover .table-primary:hover,.table-hover .table-primary:hover>td,.table-hover .table-primary:hover>th{background-color:#9fcdff}.table-secondary,.table-secondary>td,.table-secondary>th{background-color:#d6d8db}.table-secondary tbody+tbody,.table-secondary td,.table-secondary th,.table-secondary thead th{border-color:#b3b7bb}.table-hover .table-secondary:hover,.table-hover .table-secondary:hover>td,.table-hover .table-secondary:hover>th{background-color:#c8cbcf}.table-success,.table-success>td,.table-success>th{background-color:#c3e6cb}.table-success tbody+tbody,.table-success td,.table-success th,.table-success thead th{border-color:#8fd19e}.table-hover .table-success:hover,.table-hover .table-success:hover>td,.table-hover .table-success:hover>th{background-color:#b1dfbb}.table-info,.table-info>td,.table-info>th{background-color:#bee5eb}.table-info tbody+tbody,.table-info td,.table-info th,.table-info thead th{border-color:#86cfda}.table-hover .table-info:hover,.table-hover .table-info:hover>td,.table-hover .table-info:hover>th{background-color:#abdde5}.table-warning,.table-warning>td,.table-warning>th{background-color:#ffeeba}.table-warning tbody+tbody,.table-warning td,.table-warning th,.table-warning thead th{border-color:#ffdf7e}.table-hover .table-warning:hover,.table-hover .table-warning:hover>td,.table-hover .table-warning:hover>th{background-color:#ffe8a1}.table-danger,.table-danger>td,.table-danger>th{background-color:#f5c6cb}.table-danger tbody+tbody,.table-danger td,.table-danger th,.table-danger thead th{border-color:#ed969e}.table-hover .table-danger:hover,.table-hover .table-danger:hover>td,.table-hover .table-danger:hover>th{background-color:#f1b0b7}.table-light,.table-light>td,.table-light>th{background-color:#fdfdfe}.table-light tbody+tbody,.table-light td,.table-light th,.table-light thead th{border-color:#fbfcfc}.table-hover .table-light:hover,.table-hover .table-light:hover>td,.table-hover .table-light:hover>th{background-color:#ececf6}.table-dark,.table-dark>td,.table-dark>th{background-color:#c6c8ca}.table-dark tbody+tbody,.table-dark td,.table-dark th,.table-dark thead th{border-color:#95999c}.table-hover .table-dark:hover,.table-hover .table-dark:hover>td,.table-hover .table-dark:hover>th{background-color:#b9bbbe}.table-active,.table-active>td,.table-active>th,.table-hover .table-active:hover,.table-hover .table-active:hover>td,.table-hover .table-active:hover>th{background-color:rgba(0,0,0,.075)}.table .thead-dark th{color:#fff;background-color:#343a40;border-color:#454d55}.table .thead-light th{color:#495057;background-color:#e9ecef;border-color:#dee2e6}.table-dark{color:#fff;background-color:#343a40}.table-dark td,.table-dark th,.table-dark thead th{border-color:#454d55}.table-dark.table-bordered{border:0}.table-dark.table-striped tbody tr:nth-of-type(odd){background-color:hsla(0,0%,100%,.05)}.table-dark.table-hover tbody tr:hover{color:#fff;background-color:hsla(0,0%,100%,.075)}@media (max-width:575.98px){.table-responsive-sm{display:block;width:100%;overflow-x:auto;-webkit-overflow-scrolling:touch}.table-responsive-sm>.table-bordered{border:0}}@media (max-width:767.98px){.table-responsive-md{display:block;width:100%;overflow-x:auto;-webkit-overflow-scrolling:touch}.table-responsive-md>.table-bordered{border:0}}@media (max-width:991.98px){.table-responsive-lg{display:block;width:100%;overflow-x:auto;-webkit-overflow-scrolling:touch}.table-responsive-lg>.table-bordered{border:0}}@media (max-width:1199.98px){.table-responsive-xl{display:block;width:100%;overflow-x:auto;-webkit-overflow-scrolling:touch}.table-responsive-xl>.table-bordered{border:0}}.table-responsive{display:block;width:100%;overflow-x:auto;-webkit-overflow-scrolling:touch}.table-responsive>.table-bordered{border:0}.form-control{display:block;width:100%;height:calc(1.5em + .75rem + 2px);padding:.375rem .75rem;font-size:1rem;font-weight:400;line-height:1.5;color:#495057;background-color:#fff;background-clip:padding-box;border:1px solid #ced4da;border-radius:.25rem;transition:border-color .15s ease-in-out,box-shadow .15s ease-in-out}@media (prefers-reduced-motion:reduce){.form-control{transition:none}}.form-control::-ms-expand{background-color:transparent;border:0}.form-control:-moz-focusring{color:transparent;text-shadow:0 0 0 #495057}.form-control:focus{color:#495057;background-color:#fff;border-color:#80bdff;outline:0;box-shadow:0 0 0 .2rem rgba(0,123,255,.25)}.form-control::placeholder{color:#6c757d;opacity:1}.form-control:disabled,.form-control[readonly]{background-color:#e9ecef;opacity:1}input[type=date].form-control,input[type=datetime-local].form-control,input[type=month].form-control,input[type=time].form-control{appearance:none}select.form-control:focus::-ms-value{color:#495057;background-color:#fff}.form-control-file,.form-control-range{display:block;width:100%}.col-form-label{padding-top:calc(.375rem + 1px);padding-bottom:calc(.375rem + 1px);margin-bottom:0;font-size:inherit;line-height:1.5}.col-form-label-lg{padding-top:calc(.5rem + 1px);padding-bottom:calc(.5rem + 1px);font-size:1.25rem;line-height:1.5}.col-form-label-sm{padding-top:calc(.25rem + 1px);padding-bottom:calc(.25rem + 1px);font-size:.875rem;line-height:1.5}.form-control-plaintext{display:block;width:100%;padding:.375rem 0;margin-bottom:0;font-size:1rem;line-height:1.5;color:#212529;background-color:transparent;border:solid transparent;border-width:1px 0}.form-control-plaintext.form-control-lg,.form-control-plaintext.form-control-sm{padding-right:0;padding-left:0}.form-control-sm{height:calc(1.5em + .5rem + 2px);padding:.25rem .5rem;font-size:.875rem;line-height:1.5;border-radius:.2rem}.form-control-lg{height:calc(1.5em + 1rem + 2px);padding:.5rem 1rem;font-size:1.25rem;line-height:1.5;border-radius:.3rem}select.form-control[multiple],select.form-control[size],textarea.form-control{height:auto}.form-group{margin-bottom:1rem}.form-text{display:block;margin-top:.25rem}.form-row{display:flex;flex-wrap:wrap;margin-right:-5px;margin-left:-5px}.form-row>.col,.form-row>[class*=col-]{padding-right:5px;padding-left:5px}.form-check{position:relative;display:block;padding-left:1.25rem}.form-check-input{position:absolute;margin-top:.3rem;margin-left:-1.25rem}.form-check-input:disabled~.form-check-label,.form-check-input[disabled]~.form-check-label{color:#6c757d}.form-check-label{margin-bottom:0}.form-check-inline{display:inline-flex;align-items:center;padding-left:0;margin-right:.75rem}.form-check-inline .form-check-input{position:static;margin-top:0;margin-right:.3125rem;margin-left:0}.valid-feedback{display:none;width:100%;margin-top:.25rem;font-size:80%;color:#28a745}.valid-tooltip{position:absolute;top:100%;z-index:5;display:none;max-width:100%;padding:.25rem .5rem;margin-top:.1rem;font-size:.875rem;line-height:1.5;color:#fff;background-color:rgba(40,167,69,.9);border-radius:.25rem}.is-valid~.valid-feedback,.is-valid~.valid-tooltip,.was-validated :valid~.valid-feedback,.was-validated :valid~.valid-tooltip{display:block}.form-control.is-valid,.was-validated .form-control:valid{border-color:#28a745;padding-right:calc(1.5em + .75rem);background-image:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='8' height='8'%3E%3Cpath fill='%2328a745' d='M2.3 6.73L.6 4.53c-.4-1.04.46-1.4 1.1-.8l1.1 1.4 3.4-3.8c.6-.63 1.6-.27 1.2.7l-4 4.6c-.43.5-.8.4-1.1.1z'/%3E%3C/svg%3E");background-repeat:no-repeat;background-position:right calc(.375em + .1875rem) center;background-size:calc(.75em + .375rem) calc(.75em + .375rem)}.form-control.is-valid:focus,.was-validated .form-control:valid:focus{border-color:#28a745;box-shadow:0 0 0 .2rem rgba(40,167,69,.25)}.was-validated textarea.form-control:valid,textarea.form-control.is-valid{padding-right:calc(1.5em + .75rem);background-position:top calc(.375em + .1875rem) right calc(.375em + .1875rem)}.custom-select.is-valid,.was-validated .custom-select:valid{border-color:#28a745;padding-right:calc(.75em + 2.3125rem);background:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='4' height='5'%3E%3Cpath fill='%23343a40' d='M2 0L0 2h4zm0 5L0 3h4z'/%3E%3C/svg%3E") no-repeat right .75rem center/8px 10px,url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='8' height='8'%3E%3Cpath fill='%2328a745' d='M2.3 6.73L.6 4.53c-.4-1.04.46-1.4 1.1-.8l1.1 1.4 3.4-3.8c.6-.63 1.6-.27 1.2.7l-4 4.6c-.43.5-.8.4-1.1.1z'/%3E%3C/svg%3E") #fff no-repeat center right 1.75rem/calc(.75em + .375rem) calc(.75em + .375rem)}.custom-select.is-valid:focus,.was-validated .custom-select:valid:focus{border-color:#28a745;box-shadow:0 0 0 .2rem rgba(40,167,69,.25)}.form-check-input.is-valid~.form-check-label,.was-validated .form-check-input:valid~.form-check-label{color:#28a745}.form-check-input.is-valid~.valid-feedback,.form-check-input.is-valid~.valid-tooltip,.was-validated .form-check-input:valid~.valid-feedback,.was-validated .form-check-input:valid~.valid-tooltip{display:block}.custom-control-input.is-valid~.custom-control-label,.was-validated .custom-control-input:valid~.custom-control-label{color:#28a745}.custom-control-input.is-valid~.custom-control-label:before,.was-validated .custom-control-input:valid~.custom-control-label:before{border-color:#28a745}.custom-control-input.is-valid:checked~.custom-control-label:before,.was-validated .custom-control-input:valid:checked~.custom-control-label:before{border-color:#34ce57;background-color:#34ce57}.custom-control-input.is-valid:focus~.custom-control-label:before,.was-validated .custom-control-input:valid:focus~.custom-control-label:before{box-shadow:0 0 0 .2rem rgba(40,167,69,.25)}.custom-control-input.is-valid:focus:not(:checked)~.custom-control-label:before,.custom-file-input.is-valid~.custom-file-label,.was-validated .custom-control-input:valid:focus:not(:checked)~.custom-control-label:before,.was-validated .custom-file-input:valid~.custom-file-label{border-color:#28a745}.custom-file-input.is-valid:focus~.custom-file-label,.was-validated .custom-file-input:valid:focus~.custom-file-label{border-color:#28a745;box-shadow:0 0 0 .2rem rgba(40,167,69,.25)}.invalid-feedback{display:none;width:100%;margin-top:.25rem;font-size:80%;color:#dc3545}.invalid-tooltip{position:absolute;top:100%;z-index:5;display:none;max-width:100%;padding:.25rem .5rem;margin-top:.1rem;font-size:.875rem;line-height:1.5;color:#fff;background-color:rgba(220,53,69,.9);border-radius:.25rem}.is-invalid~.invalid-feedback,.is-invalid~.invalid-tooltip,.was-validated :invalid~.invalid-feedback,.was-validated :invalid~.invalid-tooltip{display:block}.form-control.is-invalid,.was-validated .form-control:invalid{border-color:#dc3545;padding-right:calc(1.5em + .75rem);background-image:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='12' height='12' fill='none' stroke='%23dc3545'%3E%3Ccircle cx='6' cy='6' r='4.5'/%3E%3Cpath stroke-linejoin='round' d='M5.8 3.6h.4L6 6.5z'/%3E%3Ccircle cx='6' cy='8.2' r='.6' fill='%23dc3545' stroke='none'/%3E%3C/svg%3E");background-repeat:no-repeat;background-position:right calc(.375em + .1875rem) center;background-size:calc(.75em + .375rem) calc(.75em + .375rem)}.form-control.is-invalid:focus,.was-validated .form-control:invalid:focus{border-color:#dc3545;box-shadow:0 0 0 .2rem rgba(220,53,69,.25)}.was-validated textarea.form-control:invalid,textarea.form-control.is-invalid{padding-right:calc(1.5em + .75rem);background-position:top calc(.375em + .1875rem) right calc(.375em + .1875rem)}.custom-select.is-invalid,.was-validated .custom-select:invalid{border-color:#dc3545;padding-right:calc(.75em + 2.3125rem);background:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='4' height='5'%3E%3Cpath fill='%23343a40' d='M2 0L0 2h4zm0 5L0 3h4z'/%3E%3C/svg%3E") no-repeat right .75rem center/8px 10px,url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='12' height='12' fill='none' stroke='%23dc3545'%3E%3Ccircle cx='6' cy='6' r='4.5'/%3E%3Cpath stroke-linejoin='round' d='M5.8 3.6h.4L6 6.5z'/%3E%3Ccircle cx='6' cy='8.2' r='.6' fill='%23dc3545' stroke='none'/%3E%3C/svg%3E") #fff no-repeat center right 1.75rem/calc(.75em + .375rem) calc(.75em + .375rem)}.custom-select.is-invalid:focus,.was-validated .custom-select:invalid:focus{border-color:#dc3545;box-shadow:0 0 0 .2rem rgba(220,53,69,.25)}.form-check-input.is-invalid~.form-check-label,.was-validated .form-check-input:invalid~.form-check-label{color:#dc3545}.form-check-input.is-invalid~.invalid-feedback,.form-check-input.is-invalid~.invalid-tooltip,.was-validated .form-check-input:invalid~.invalid-feedback,.was-validated .form-check-input:invalid~.invalid-tooltip{display:block}.custom-control-input.is-invalid~.custom-control-label,.was-validated .custom-control-input:invalid~.custom-control-label{color:#dc3545}.custom-control-input.is-invalid~.custom-control-label:before,.was-validated .custom-control-input:invalid~.custom-control-label:before{border-color:#dc3545}.custom-control-input.is-invalid:checked~.custom-control-label:before,.was-validated .custom-control-input:invalid:checked~.custom-control-label:before{border-color:#e4606d;background-color:#e4606d}.custom-control-input.is-invalid:focus~.custom-control-label:before,.was-validated .custom-control-input:invalid:focus~.custom-control-label:before{box-shadow:0 0 0 .2rem rgba(220,53,69,.25)}.custom-control-input.is-invalid:focus:not(:checked)~.custom-control-label:before,.custom-file-input.is-invalid~.custom-file-label,.was-validated .custom-control-input:invalid:focus:not(:checked)~.custom-control-label:before,.was-validated .custom-file-input:invalid~.custom-file-label{border-color:#dc3545}.custom-file-input.is-invalid:focus~.custom-file-label,.was-validated .custom-file-input:invalid:focus~.custom-file-label{border-color:#dc3545;box-shadow:0 0 0 .2rem rgba(220,53,69,.25)}.form-inline{display:flex;flex-flow:row wrap;align-items:center}.form-inline .form-check{width:100%}@media (min-width:576px){.form-inline label{justify-content:center}.form-inline .form-group,.form-inline label{display:flex;align-items:center;margin-bottom:0}.form-inline .form-group{flex:0 0 auto;flex-flow:row wrap}.form-inline .form-control{display:inline-block;width:auto;vertical-align:middle}.form-inline .form-control-plaintext{display:inline-block}.form-inline .custom-select,.form-inline .input-group{width:auto}.form-inline .form-check{display:flex;align-items:center;justify-content:center;width:auto;padding-left:0}.form-inline .form-check-input{position:relative;flex-shrink:0;margin-top:0;margin-right:.25rem;margin-left:0}.form-inline .custom-control{align-items:center;justify-content:center}.form-inline .custom-control-label{margin-bottom:0}}.btn{display:inline-block;font-weight:400;color:#212529;text-align:center;vertical-align:middle;user-select:none;background-color:transparent;border:1px solid transparent;padding:.375rem .75rem;font-size:1rem;line-height:1.5;border-radius:.25rem;transition:color .15s ease-in-out,background-color .15s ease-in-out,border-color .15s ease-in-out,box-shadow .15s ease-in-out}@media (prefers-reduced-motion:reduce){.btn{transition:none}}.btn:hover{color:#212529;text-decoration:none}.btn.focus,.btn:focus{outline:0;box-shadow:0 0 0 .2rem rgba(0,123,255,.25)}.btn.disabled,.btn:disabled{opacity:.65}.btn:not(:disabled):not(.disabled){cursor:pointer}a.btn.disabled,fieldset:disabled a.btn{pointer-events:none}.btn-primary{color:#fff;background-color:#007bff;border-color:#007bff}.btn-primary.focus,.btn-primary:focus,.btn-primary:hover{color:#fff;background-color:#0069d9;border-color:#0062cc}.btn-primary.focus,.btn-primary:focus{box-shadow:0 0 0 .2rem rgba(38,143,255,.5)}.btn-primary.disabled,.btn-primary:disabled{color:#fff;background-color:#007bff;border-color:#007bff}.btn-primary:not(:disabled):not(.disabled).active,.btn-primary:not(:disabled):not(.disabled):active,.show>.btn-primary.dropdown-toggle{color:#fff;background-color:#0062cc;border-color:#005cbf}.btn-primary:not(:disabled):not(.disabled).active:focus,.btn-primary:not(:disabled):not(.disabled):active:focus,.show>.btn-primary.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(38,143,255,.5)}.btn-secondary{color:#fff;background-color:#6c757d;border-color:#6c757d}.btn-secondary.focus,.btn-secondary:focus,.btn-secondary:hover{color:#fff;background-color:#5a6268;border-color:#545b62}.btn-secondary.focus,.btn-secondary:focus{box-shadow:0 0 0 .2rem rgba(130,138,145,.5)}.btn-secondary.disabled,.btn-secondary:disabled{color:#fff;background-color:#6c757d;border-color:#6c757d}.btn-secondary:not(:disabled):not(.disabled).active,.btn-secondary:not(:disabled):not(.disabled):active,.show>.btn-secondary.dropdown-toggle{color:#fff;background-color:#545b62;border-color:#4e555b}.btn-secondary:not(:disabled):not(.disabled).active:focus,.btn-secondary:not(:disabled):not(.disabled):active:focus,.show>.btn-secondary.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(130,138,145,.5)}.btn-success{color:#fff;background-color:#28a745;border-color:#28a745}.btn-success.focus,.btn-success:focus,.btn-success:hover{color:#fff;background-color:#218838;border-color:#1e7e34}.btn-success.focus,.btn-success:focus{box-shadow:0 0 0 .2rem rgba(72,180,97,.5)}.btn-success.disabled,.btn-success:disabled{color:#fff;background-color:#28a745;border-color:#28a745}.btn-success:not(:disabled):not(.disabled).active,.btn-success:not(:disabled):not(.disabled):active,.show>.btn-success.dropdown-toggle{color:#fff;background-color:#1e7e34;border-color:#1c7430}.btn-success:not(:disabled):not(.disabled).active:focus,.btn-success:not(:disabled):not(.disabled):active:focus,.show>.btn-success.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(72,180,97,.5)}.btn-info{color:#fff;background-color:#17a2b8;border-color:#17a2b8}.btn-info.focus,.btn-info:focus,.btn-info:hover{color:#fff;background-color:#138496;border-color:#117a8b}.btn-info.focus,.btn-info:focus{box-shadow:0 0 0 .2rem rgba(58,176,195,.5)}.btn-info.disabled,.btn-info:disabled{color:#fff;background-color:#17a2b8;border-color:#17a2b8}.btn-info:not(:disabled):not(.disabled).active,.btn-info:not(:disabled):not(.disabled):active,.show>.btn-info.dropdown-toggle{color:#fff;background-color:#117a8b;border-color:#10707f}.btn-info:not(:disabled):not(.disabled).active:focus,.btn-info:not(:disabled):not(.disabled):active:focus,.show>.btn-info.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(58,176,195,.5)}.btn-warning{color:#212529;background-color:#ffc107;border-color:#ffc107}.btn-warning.focus,.btn-warning:focus,.btn-warning:hover{color:#212529;background-color:#e0a800;border-color:#d39e00}.btn-warning.focus,.btn-warning:focus{box-shadow:0 0 0 .2rem rgba(222,170,12,.5)}.btn-warning.disabled,.btn-warning:disabled{color:#212529;background-color:#ffc107;border-color:#ffc107}.btn-warning:not(:disabled):not(.disabled).active,.btn-warning:not(:disabled):not(.disabled):active,.show>.btn-warning.dropdown-toggle{color:#212529;background-color:#d39e00;border-color:#c69500}.btn-warning:not(:disabled):not(.disabled).active:focus,.btn-warning:not(:disabled):not(.disabled):active:focus,.show>.btn-warning.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(222,170,12,.5)}.btn-danger{color:#fff;background-color:#dc3545;border-color:#dc3545}.btn-danger.focus,.btn-danger:focus,.btn-danger:hover{color:#fff;background-color:#c82333;border-color:#bd2130}.btn-danger.focus,.btn-danger:focus{box-shadow:0 0 0 .2rem rgba(225,83,97,.5)}.btn-danger.disabled,.btn-danger:disabled{color:#fff;background-color:#dc3545;border-color:#dc3545}.btn-danger:not(:disabled):not(.disabled).active,.btn-danger:not(:disabled):not(.disabled):active,.show>.btn-danger.dropdown-toggle{color:#fff;background-color:#bd2130;border-color:#b21f2d}.btn-danger:not(:disabled):not(.disabled).active:focus,.btn-danger:not(:disabled):not(.disabled):active:focus,.show>.btn-danger.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(225,83,97,.5)}.btn-light{color:#212529;background-color:#f8f9fa;border-color:#f8f9fa}.btn-light.focus,.btn-light:focus,.btn-light:hover{color:#212529;background-color:#e2e6ea;border-color:#dae0e5}.btn-light.focus,.btn-light:focus{box-shadow:0 0 0 .2rem rgba(216,217,219,.5)}.btn-light.disabled,.btn-light:disabled{color:#212529;background-color:#f8f9fa;border-color:#f8f9fa}.btn-light:not(:disabled):not(.disabled).active,.btn-light:not(:disabled):not(.disabled):active,.show>.btn-light.dropdown-toggle{color:#212529;background-color:#dae0e5;border-color:#d3d9df}.btn-light:not(:disabled):not(.disabled).active:focus,.btn-light:not(:disabled):not(.disabled):active:focus,.show>.btn-light.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(216,217,219,.5)}.btn-dark{color:#fff;background-color:#343a40;border-color:#343a40}.btn-dark.focus,.btn-dark:focus,.btn-dark:hover{color:#fff;background-color:#23272b;border-color:#1d2124}.btn-dark.focus,.btn-dark:focus{box-shadow:0 0 0 .2rem rgba(82,88,93,.5)}.btn-dark.disabled,.btn-dark:disabled{color:#fff;background-color:#343a40;border-color:#343a40}.btn-dark:not(:disabled):not(.disabled).active,.btn-dark:not(:disabled):not(.disabled):active,.show>.btn-dark.dropdown-toggle{color:#fff;background-color:#1d2124;border-color:#171a1d}.btn-dark:not(:disabled):not(.disabled).active:focus,.btn-dark:not(:disabled):not(.disabled):active:focus,.show>.btn-dark.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(82,88,93,.5)}.btn-outline-primary{color:#007bff;border-color:#007bff}.btn-outline-primary:hover{color:#fff;background-color:#007bff;border-color:#007bff}.btn-outline-primary.focus,.btn-outline-primary:focus{box-shadow:0 0 0 .2rem rgba(0,123,255,.5)}.btn-outline-primary.disabled,.btn-outline-primary:disabled{color:#007bff;background-color:transparent}.btn-outline-primary:not(:disabled):not(.disabled).active,.btn-outline-primary:not(:disabled):not(.disabled):active,.show>.btn-outline-primary.dropdown-toggle{color:#fff;background-color:#007bff;border-color:#007bff}.btn-outline-primary:not(:disabled):not(.disabled).active:focus,.btn-outline-primary:not(:disabled):not(.disabled):active:focus,.show>.btn-outline-primary.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(0,123,255,.5)}.btn-outline-secondary{color:#6c757d;border-color:#6c757d}.btn-outline-secondary:hover{color:#fff;background-color:#6c757d;border-color:#6c757d}.btn-outline-secondary.focus,.btn-outline-secondary:focus{box-shadow:0 0 0 .2rem rgba(108,117,125,.5)}.btn-outline-secondary.disabled,.btn-outline-secondary:disabled{color:#6c757d;background-color:transparent}.btn-outline-secondary:not(:disabled):not(.disabled).active,.btn-outline-secondary:not(:disabled):not(.disabled):active,.show>.btn-outline-secondary.dropdown-toggle{color:#fff;background-color:#6c757d;border-color:#6c757d}.btn-outline-secondary:not(:disabled):not(.disabled).active:focus,.btn-outline-secondary:not(:disabled):not(.disabled):active:focus,.show>.btn-outline-secondary.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(108,117,125,.5)}.btn-outline-success{color:#28a745;border-color:#28a745}.btn-outline-success:hover{color:#fff;background-color:#28a745;border-color:#28a745}.btn-outline-success.focus,.btn-outline-success:focus{box-shadow:0 0 0 .2rem rgba(40,167,69,.5)}.btn-outline-success.disabled,.btn-outline-success:disabled{color:#28a745;background-color:transparent}.btn-outline-success:not(:disabled):not(.disabled).active,.btn-outline-success:not(:disabled):not(.disabled):active,.show>.btn-outline-success.dropdown-toggle{color:#fff;background-color:#28a745;border-color:#28a745}.btn-outline-success:not(:disabled):not(.disabled).active:focus,.btn-outline-success:not(:disabled):not(.disabled):active:focus,.show>.btn-outline-success.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(40,167,69,.5)}.btn-outline-info{color:#17a2b8;border-color:#17a2b8}.btn-outline-info:hover{color:#fff;background-color:#17a2b8;border-color:#17a2b8}.btn-outline-info.focus,.btn-outline-info:focus{box-shadow:0 0 0 .2rem rgba(23,162,184,.5)}.btn-outline-info.disabled,.btn-outline-info:disabled{color:#17a2b8;background-color:transparent}.btn-outline-info:not(:disabled):not(.disabled).active,.btn-outline-info:not(:disabled):not(.disabled):active,.show>.btn-outline-info.dropdown-toggle{color:#fff;background-color:#17a2b8;border-color:#17a2b8}.btn-outline-info:not(:disabled):not(.disabled).active:focus,.btn-outline-info:not(:disabled):not(.disabled):active:focus,.show>.btn-outline-info.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(23,162,184,.5)}.btn-outline-warning{color:#ffc107;border-color:#ffc107}.btn-outline-warning:hover{color:#212529;background-color:#ffc107;border-color:#ffc107}.btn-outline-warning.focus,.btn-outline-warning:focus{box-shadow:0 0 0 .2rem rgba(255,193,7,.5)}.btn-outline-warning.disabled,.btn-outline-warning:disabled{color:#ffc107;background-color:transparent}.btn-outline-warning:not(:disabled):not(.disabled).active,.btn-outline-warning:not(:disabled):not(.disabled):active,.show>.btn-outline-warning.dropdown-toggle{color:#212529;background-color:#ffc107;border-color:#ffc107}.btn-outline-warning:not(:disabled):not(.disabled).active:focus,.btn-outline-warning:not(:disabled):not(.disabled):active:focus,.show>.btn-outline-warning.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(255,193,7,.5)}.btn-outline-danger{color:#dc3545;border-color:#dc3545}.btn-outline-danger:hover{color:#fff;background-color:#dc3545;border-color:#dc3545}.btn-outline-danger.focus,.btn-outline-danger:focus{box-shadow:0 0 0 .2rem rgba(220,53,69,.5)}.btn-outline-danger.disabled,.btn-outline-danger:disabled{color:#dc3545;background-color:transparent}.btn-outline-danger:not(:disabled):not(.disabled).active,.btn-outline-danger:not(:disabled):not(.disabled):active,.show>.btn-outline-danger.dropdown-toggle{color:#fff;background-color:#dc3545;border-color:#dc3545}.btn-outline-danger:not(:disabled):not(.disabled).active:focus,.btn-outline-danger:not(:disabled):not(.disabled):active:focus,.show>.btn-outline-danger.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(220,53,69,.5)}.btn-outline-light{color:#f8f9fa;border-color:#f8f9fa}.btn-outline-light:hover{color:#212529;background-color:#f8f9fa;border-color:#f8f9fa}.btn-outline-light.focus,.btn-outline-light:focus{box-shadow:0 0 0 .2rem rgba(248,249,250,.5)}.btn-outline-light.disabled,.btn-outline-light:disabled{color:#f8f9fa;background-color:transparent}.btn-outline-light:not(:disabled):not(.disabled).active,.btn-outline-light:not(:disabled):not(.disabled):active,.show>.btn-outline-light.dropdown-toggle{color:#212529;background-color:#f8f9fa;border-color:#f8f9fa}.btn-outline-light:not(:disabled):not(.disabled).active:focus,.btn-outline-light:not(:disabled):not(.disabled):active:focus,.show>.btn-outline-light.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(248,249,250,.5)}.btn-outline-dark{color:#343a40;border-color:#343a40}.btn-outline-dark:hover{color:#fff;background-color:#343a40;border-color:#343a40}.btn-outline-dark.focus,.btn-outline-dark:focus{box-shadow:0 0 0 .2rem rgba(52,58,64,.5)}.btn-outline-dark.disabled,.btn-outline-dark:disabled{color:#343a40;background-color:transparent}.btn-outline-dark:not(:disabled):not(.disabled).active,.btn-outline-dark:not(:disabled):not(.disabled):active,.show>.btn-outline-dark.dropdown-toggle{color:#fff;background-color:#343a40;border-color:#343a40}.btn-outline-dark:not(:disabled):not(.disabled).active:focus,.btn-outline-dark:not(:disabled):not(.disabled):active:focus,.show>.btn-outline-dark.dropdown-toggle:focus{box-shadow:0 0 0 .2rem rgba(52,58,64,.5)}.btn-link{font-weight:400;color:#007bff;text-decoration:none}.btn-link:hover{color:#0056b3}.btn-link.focus,.btn-link:focus,.btn-link:hover{text-decoration:underline}.btn-link.disabled,.btn-link:disabled{color:#6c757d;pointer-events:none}.btn-group-lg>.btn,.btn-lg{padding:.5rem 1rem;font-size:1.25rem;line-height:1.5;border-radius:.3rem}.btn-group-sm>.btn,.btn-sm{padding:.25rem .5rem;font-size:.875rem;line-height:1.5;border-radius:.2rem}.btn-block{display:block;width:100%}.btn-block+.btn-block{margin-top:.5rem}input[type=button].btn-block,input[type=reset].btn-block,input[type=submit].btn-block{width:100%}.fade{transition:opacity .15s linear}@media (prefers-reduced-motion:reduce){.fade{transition:none}}.fade:not(.show){opacity:0}.collapse:not(.show){display:none}.collapsing{position:relative;height:0;overflow:hidden;transition:height .35s ease}@media (prefers-reduced-motion:reduce){.collapsing{transition:none}}.dropdown,.dropleft,.dropright,.dropup{position:relative}.dropdown-toggle{white-space:nowrap}.dropdown-toggle:after{display:inline-block;margin-left:.255em;vertical-align:.255em;content:"";border-top:.3em solid;border-right:.3em solid transparent;border-bottom:0;border-left:.3em solid transparent}.dropdown-toggle:empty:after{margin-left:0}.dropdown-menu{position:absolute;top:100%;left:0;z-index:1000;display:none;float:left;min-width:10rem;padding:.5rem 0;margin:.125rem 0 0;font-size:1rem;color:#212529;text-align:left;list-style:none;background-color:#fff;background-clip:padding-box;border:1px solid rgba(0,0,0,.15);border-radius:.25rem}.dropdown-menu-left{right:auto;left:0}.dropdown-menu-right{right:0;left:auto}@media (min-width:576px){.dropdown-menu-sm-left{right:auto;left:0}.dropdown-menu-sm-right{right:0;left:auto}}@media (min-width:768px){.dropdown-menu-md-left{right:auto;left:0}.dropdown-menu-md-right{right:0;left:auto}}@media (min-width:992px){.dropdown-menu-lg-left{right:auto;left:0}.dropdown-menu-lg-right{right:0;left:auto}}@media (min-width:1200px){.dropdown-menu-xl-left{right:auto;left:0}.dropdown-menu-xl-right{right:0;left:auto}}.dropup .dropdown-menu{top:auto;bottom:100%;margin-top:0;margin-bottom:.125rem}.dropup .dropdown-toggle:after{display:inline-block;margin-left:.255em;vertical-align:.255em;content:"";border-top:0;border-right:.3em solid transparent;border-bottom:.3em solid;border-left:.3em solid transparent}.dropup .dropdown-toggle:empty:after{margin-left:0}.dropright .dropdown-menu{top:0;right:auto;left:100%;margin-top:0;margin-left:.125rem}.dropright .dropdown-toggle:after{display:inline-block;margin-left:.255em;vertical-align:.255em;content:"";border-top:.3em solid transparent;border-right:0;border-bottom:.3em solid transparent;border-left:.3em solid}.dropright .dropdown-toggle:empty:after{margin-left:0}.dropright .dropdown-toggle:after{vertical-align:0}.dropleft .dropdown-menu{top:0;right:100%;left:auto;margin-top:0;margin-right:.125rem}.dropleft .dropdown-toggle:after{display:inline-block;margin-left:.255em;vertical-align:.255em;content:"";display:none}.dropleft .dropdown-toggle:before{display:inline-block;margin-right:.255em;vertical-align:.255em;content:"";border-top:.3em solid transparent;border-right:.3em solid;border-bottom:.3em solid transparent}.dropleft .dropdown-toggle:empty:after{margin-left:0}.dropleft .dropdown-toggle:before{vertical-align:0}.dropdown-menu[x-placement^=bottom],.dropdown-menu[x-placement^=left],.dropdown-menu[x-placement^=right],.dropdown-menu[x-placement^=top]{right:auto;bottom:auto}.dropdown-divider{height:0;margin:.5rem 0;overflow:hidden;border-top:1px solid #e9ecef}.dropdown-item{display:block;width:100%;padding:.25rem 1.5rem;clear:both;font-weight:400;color:#212529;text-align:inherit;white-space:nowrap;background-color:transparent;border:0}.dropdown-item:focus,.dropdown-item:hover{color:#16181b;text-decoration:none;background-color:#f8f9fa}.dropdown-item.active,.dropdown-item:active{color:#fff;text-decoration:none;background-color:#007bff}.dropdown-item.disabled,.dropdown-item:disabled{color:#6c757d;pointer-events:none;background-color:transparent}.dropdown-menu.show{display:block}.dropdown-header{display:block;padding:.5rem 1.5rem;margin-bottom:0;font-size:.875rem;color:#6c757d;white-space:nowrap}.dropdown-item-text{display:block;padding:.25rem 1.5rem;color:#212529}.btn-group,.btn-group-vertical{position:relative;display:inline-flex;vertical-align:middle}.btn-group-vertical>.btn,.btn-group>.btn{position:relative;flex:1 1 auto}.btn-group-vertical>.btn.active,.btn-group-vertical>.btn:active,.btn-group-vertical>.btn:focus,.btn-group-vertical>.btn:hover,.btn-group>.btn.active,.btn-group>.btn:active,.btn-group>.btn:focus,.btn-group>.btn:hover{z-index:1}.btn-toolbar{display:flex;flex-wrap:wrap;justify-content:flex-start}.btn-toolbar .input-group{width:auto}.btn-group>.btn-group:not(:first-child),.btn-group>.btn:not(:first-child){margin-left:-1px}.btn-group>.btn-group:not(:last-child)>.btn,.btn-group>.btn:not(:last-child):not(.dropdown-toggle){border-top-right-radius:0;border-bottom-right-radius:0}.btn-group>.btn-group:not(:first-child)>.btn,.btn-group>.btn:not(:first-child){border-top-left-radius:0;border-bottom-left-radius:0}.dropdown-toggle-split{padding-right:.5625rem;padding-left:.5625rem}.dropdown-toggle-split:after,.dropright .dropdown-toggle-split:after,.dropup .dropdown-toggle-split:after{margin-left:0}.dropleft .dropdown-toggle-split:before{margin-right:0}.btn-group-sm>.btn+.dropdown-toggle-split,.btn-sm+.dropdown-toggle-split{padding-right:.375rem;padding-left:.375rem}.btn-group-lg>.btn+.dropdown-toggle-split,.btn-lg+.dropdown-toggle-split{padding-right:.75rem;padding-left:.75rem}.btn-group-vertical{flex-direction:column;align-items:flex-start;justify-content:center}.btn-group-vertical>.btn,.btn-group-vertical>.btn-group{width:100%}.btn-group-vertical>.btn-group:not(:first-child),.btn-group-vertical>.btn:not(:first-child){margin-top:-1px}.btn-group-vertical>.btn-group:not(:last-child)>.btn,.btn-group-vertical>.btn:not(:last-child):not(.dropdown-toggle){border-bottom-right-radius:0;border-bottom-left-radius:0}.btn-group-vertical>.btn-group:not(:first-child)>.btn,.btn-group-vertical>.btn:not(:first-child){border-top-left-radius:0;border-top-right-radius:0}.btn-group-toggle>.btn,.btn-group-toggle>.btn-group>.btn{margin-bottom:0}.btn-group-toggle>.btn-group>.btn input[type=checkbox],.btn-group-toggle>.btn-group>.btn input[type=radio],.btn-group-toggle>.btn input[type=checkbox],.btn-group-toggle>.btn input[type=radio]{position:absolute;clip:rect(0,0,0,0);pointer-events:none}.input-group{position:relative;display:flex;flex-wrap:wrap;align-items:stretch;width:100%}.input-group>.custom-file,.input-group>.custom-select,.input-group>.form-control,.input-group>.form-control-plaintext{position:relative;flex:1 1 auto;width:1%;min-width:0;margin-bottom:0}.input-group>.custom-file+.custom-file,.input-group>.custom-file+.custom-select,.input-group>.custom-file+.form-control,.input-group>.custom-select+.custom-file,.input-group>.custom-select+.custom-select,.input-group>.custom-select+.form-control,.input-group>.form-control+.custom-file,.input-group>.form-control+.custom-select,.input-group>.form-control+.form-control,.input-group>.form-control-plaintext+.custom-file,.input-group>.form-control-plaintext+.custom-select,.input-group>.form-control-plaintext+.form-control{margin-left:-1px}.input-group>.custom-file .custom-file-input:focus~.custom-file-label,.input-group>.custom-select:focus,.input-group>.form-control:focus{z-index:3}.input-group>.custom-file .custom-file-input:focus{z-index:4}.input-group>.custom-select:not(:last-child),.input-group>.form-control:not(:last-child){border-top-right-radius:0;border-bottom-right-radius:0}.input-group>.custom-select:not(:first-child),.input-group>.form-control:not(:first-child){border-top-left-radius:0;border-bottom-left-radius:0}.input-group>.custom-file{display:flex;align-items:center}.input-group>.custom-file:not(:last-child) .custom-file-label,.input-group>.custom-file:not(:last-child) .custom-file-label:after{border-top-right-radius:0;border-bottom-right-radius:0}.input-group>.custom-file:not(:first-child) .custom-file-label{border-top-left-radius:0;border-bottom-left-radius:0}.input-group-append,.input-group-prepend{display:flex}.input-group-append .btn,.input-group-prepend .btn{position:relative;z-index:2}.input-group-append .btn:focus,.input-group-prepend .btn:focus{z-index:3}.input-group-append .btn+.btn,.input-group-append .btn+.input-group-text,.input-group-append .input-group-text+.btn,.input-group-append .input-group-text+.input-group-text,.input-group-prepend .btn+.btn,.input-group-prepend .btn+.input-group-text,.input-group-prepend .input-group-text+.btn,.input-group-prepend .input-group-text+.input-group-text{margin-left:-1px}.input-group-prepend{margin-right:-1px}.input-group-append{margin-left:-1px}.input-group-text{display:flex;align-items:center;padding:.375rem .75rem;margin-bottom:0;font-size:1rem;font-weight:400;line-height:1.5;color:#495057;text-align:center;white-space:nowrap;background-color:#e9ecef;border:1px solid #ced4da;border-radius:.25rem}.input-group-text input[type=checkbox],.input-group-text input[type=radio]{margin-top:0}.input-group-lg>.custom-select,.input-group-lg>.form-control:not(textarea){height:calc(1.5em + 1rem + 2px)}.input-group-lg>.custom-select,.input-group-lg>.form-control,.input-group-lg>.input-group-append>.btn,.input-group-lg>.input-group-append>.input-group-text,.input-group-lg>.input-group-prepend>.btn,.input-group-lg>.input-group-prepend>.input-group-text{padding:.5rem 1rem;font-size:1.25rem;line-height:1.5;border-radius:.3rem}.input-group-sm>.custom-select,.input-group-sm>.form-control:not(textarea){height:calc(1.5em + .5rem + 2px)}.input-group-sm>.custom-select,.input-group-sm>.form-control,.input-group-sm>.input-group-append>.btn,.input-group-sm>.input-group-append>.input-group-text,.input-group-sm>.input-group-prepend>.btn,.input-group-sm>.input-group-prepend>.input-group-text{padding:.25rem .5rem;font-size:.875rem;line-height:1.5;border-radius:.2rem}.input-group-lg>.custom-select,.input-group-sm>.custom-select{padding-right:1.75rem}.input-group>.input-group-append:last-child>.btn:not(:last-child):not(.dropdown-toggle),.input-group>.input-group-append:last-child>.input-group-text:not(:last-child),.input-group>.input-group-append:not(:last-child)>.btn,.input-group>.input-group-append:not(:last-child)>.input-group-text,.input-group>.input-group-prepend>.btn,.input-group>.input-group-prepend>.input-group-text{border-top-right-radius:0;border-bottom-right-radius:0}.input-group>.input-group-append>.btn,.input-group>.input-group-append>.input-group-text,.input-group>.input-group-prepend:first-child>.btn:not(:first-child),.input-group>.input-group-prepend:first-child>.input-group-text:not(:first-child),.input-group>.input-group-prepend:not(:first-child)>.btn,.input-group>.input-group-prepend:not(:first-child)>.input-group-text{border-top-left-radius:0;border-bottom-left-radius:0}.custom-control{position:relative;display:block;min-height:1.5rem;padding-left:1.5rem}.custom-control-inline{display:inline-flex;margin-right:1rem}.custom-control-input{position:absolute;left:0;z-index:-1;width:1rem;height:1.25rem;opacity:0}.custom-control-input:checked~.custom-control-label:before{color:#fff;border-color:#007bff;background-color:#007bff}.custom-control-input:focus~.custom-control-label:before{box-shadow:0 0 0 .2rem rgba(0,123,255,.25)}.custom-control-input:focus:not(:checked)~.custom-control-label:before{border-color:#80bdff}.custom-control-input:not(:disabled):active~.custom-control-label:before{color:#fff;background-color:#b3d7ff;border-color:#b3d7ff}.custom-control-input:disabled~.custom-control-label,.custom-control-input[disabled]~.custom-control-label{color:#6c757d}.custom-control-input:disabled~.custom-control-label:before,.custom-control-input[disabled]~.custom-control-label:before{background-color:#e9ecef}.custom-control-label{position:relative;margin-bottom:0;vertical-align:top}.custom-control-label:before{pointer-events:none;background-color:#fff;border:1px solid #adb5bd}.custom-control-label:after,.custom-control-label:before{position:absolute;top:.25rem;left:-1.5rem;display:block;width:1rem;height:1rem;content:""}.custom-control-label:after{background:no-repeat 50%/50% 50%}.custom-checkbox .custom-control-label:before{border-radius:.25rem}.custom-checkbox .custom-control-input:checked~.custom-control-label:after{background-image:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='8' height='8'%3E%3Cpath fill='%23fff' d='M6.564.75l-3.59 3.612-1.538-1.55L0 4.26l2.974 2.99L8 2.193z'/%3E%3C/svg%3E")}.custom-checkbox .custom-control-input:indeterminate~.custom-control-label:before{border-color:#007bff;background-color:#007bff}.custom-checkbox .custom-control-input:indeterminate~.custom-control-label:after{background-image:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='4' height='4'%3E%3Cpath stroke='%23fff' d='M0 2h4'/%3E%3C/svg%3E")}.custom-checkbox .custom-control-input:disabled:checked~.custom-control-label:before{background-color:rgba(0,123,255,.5)}.custom-checkbox .custom-control-input:disabled:indeterminate~.custom-control-label:before{background-color:rgba(0,123,255,.5)}.custom-radio .custom-control-label:before{border-radius:50%}.custom-radio .custom-control-input:checked~.custom-control-label:after{background-image:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='12' height='12' viewBox='-4 -4 8 8'%3E%3Ccircle r='3' fill='%23fff'/%3E%3C/svg%3E")}.custom-radio .custom-control-input:disabled:checked~.custom-control-label:before{background-color:rgba(0,123,255,.5)}.custom-switch{padding-left:2.25rem}.custom-switch .custom-control-label:before{left:-2.25rem;width:1.75rem;pointer-events:all;border-radius:.5rem}.custom-switch .custom-control-label:after{top:calc(.25rem + 2px);left:calc(-2.25rem + 2px);width:calc(1rem - 4px);height:calc(1rem - 4px);background-color:#adb5bd;border-radius:.5rem;transition:transform .15s ease-in-out,background-color .15s ease-in-out,border-color .15s ease-in-out,box-shadow .15s ease-in-out}@media (prefers-reduced-motion:reduce){.custom-switch .custom-control-label:after{transition:none}}.custom-switch .custom-control-input:checked~.custom-control-label:after{background-color:#fff;transform:translateX(.75rem)}.custom-switch .custom-control-input:disabled:checked~.custom-control-label:before{background-color:rgba(0,123,255,.5)}.custom-select{display:inline-block;width:100%;height:calc(1.5em + .75rem + 2px);padding:.375rem 1.75rem .375rem .75rem;font-size:1rem;font-weight:400;line-height:1.5;color:#495057;vertical-align:middle;background:#fff url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='4' height='5'%3E%3Cpath fill='%23343a40' d='M2 0L0 2h4zm0 5L0 3h4z'/%3E%3C/svg%3E") no-repeat right .75rem center/8px 10px;border:1px solid #ced4da;border-radius:.25rem;appearance:none}.custom-select:focus{border-color:#80bdff;outline:0;box-shadow:0 0 0 .2rem rgba(0,123,255,.25)}.custom-select:focus::-ms-value{color:#495057;background-color:#fff}.custom-select[multiple],.custom-select[size]:not([size="1"]){height:auto;padding-right:.75rem;background-image:none}.custom-select:disabled{color:#6c757d;background-color:#e9ecef}.custom-select::-ms-expand{display:none}.custom-select:-moz-focusring{color:transparent;text-shadow:0 0 0 #495057}.custom-select-sm{height:calc(1.5em + .5rem + 2px);padding-top:.25rem;padding-bottom:.25rem;padding-left:.5rem;font-size:.875rem}.custom-select-lg{height:calc(1.5em + 1rem + 2px);padding-top:.5rem;padding-bottom:.5rem;padding-left:1rem;font-size:1.25rem}.custom-file{display:inline-block;margin-bottom:0}.custom-file,.custom-file-input{position:relative;width:100%;height:calc(1.5em + .75rem + 2px)}.custom-file-input{z-index:2;margin:0;opacity:0}.custom-file-input:focus~.custom-file-label{border-color:#80bdff;box-shadow:0 0 0 .2rem rgba(0,123,255,.25)}.custom-file-input:disabled~.custom-file-label,.custom-file-input[disabled]~.custom-file-label{background-color:#e9ecef}.custom-file-input:lang(en)~.custom-file-label:after{content:"Browse"}.custom-file-input~.custom-file-label[data-browse]:after{content:attr(data-browse)}.custom-file-label{left:0;z-index:1;height:calc(1.5em + .75rem + 2px);font-weight:400;background-color:#fff;border:1px solid #ced4da;border-radius:.25rem}.custom-file-label,.custom-file-label:after{position:absolute;top:0;right:0;padding:.375rem .75rem;line-height:1.5;color:#495057}.custom-file-label:after{bottom:0;z-index:3;display:block;height:calc(1.5em + .75rem);content:"Browse";background-color:#e9ecef;border-left:inherit;border-radius:0 .25rem .25rem 0}.custom-range{width:100%;height:1.4rem;padding:0;background-color:transparent;appearance:none}.custom-range:focus{outline:none}.custom-range:focus::-webkit-slider-thumb{box-shadow:0 0 0 1px #fff,0 0 0 .2rem rgba(0,123,255,.25)}.custom-range:focus::-moz-range-thumb{box-shadow:0 0 0 1px #fff,0 0 0 .2rem rgba(0,123,255,.25)}.custom-range:focus::-ms-thumb{box-shadow:0 0 0 1px #fff,0 0 0 .2rem rgba(0,123,255,.25)}.custom-range::-moz-focus-outer{border:0}.custom-range::-webkit-slider-thumb{width:1rem;height:1rem;margin-top:-.25rem;background-color:#007bff;border:0;border-radius:1rem;transition:background-color .15s ease-in-out,border-color .15s ease-in-out,box-shadow .15s ease-in-out;appearance:none}@media (prefers-reduced-motion:reduce){.custom-range::-webkit-slider-thumb{transition:none}}.custom-range::-webkit-slider-thumb:active{background-color:#b3d7ff}.custom-range::-webkit-slider-runnable-track{width:100%;height:.5rem;color:transparent;cursor:pointer;background-color:#dee2e6;border-color:transparent;border-radius:1rem}.custom-range::-moz-range-thumb{width:1rem;height:1rem;background-color:#007bff;border:0;border-radius:1rem;transition:background-color .15s ease-in-out,border-color .15s ease-in-out,box-shadow .15s ease-in-out;appearance:none}@media (prefers-reduced-motion:reduce){.custom-range::-moz-range-thumb{transition:none}}.custom-range::-moz-range-thumb:active{background-color:#b3d7ff}.custom-range::-moz-range-track{width:100%;height:.5rem;color:transparent;cursor:pointer;background-color:#dee2e6;border-color:transparent;border-radius:1rem}.custom-range::-ms-thumb{width:1rem;height:1rem;margin-top:0;margin-right:.2rem;margin-left:.2rem;background-color:#007bff;border:0;border-radius:1rem;transition:background-color .15s ease-in-out,border-color .15s ease-in-out,box-shadow .15s ease-in-out;appearance:none}@media (prefers-reduced-motion:reduce){.custom-range::-ms-thumb{transition:none}}.custom-range::-ms-thumb:active{background-color:#b3d7ff}.custom-range::-ms-track{width:100%;height:.5rem;color:transparent;cursor:pointer;background-color:transparent;border-color:transparent;border-width:.5rem}.custom-range::-ms-fill-lower,.custom-range::-ms-fill-upper{background-color:#dee2e6;border-radius:1rem}.custom-range::-ms-fill-upper{margin-right:15px}.custom-range:disabled::-webkit-slider-thumb{background-color:#adb5bd}.custom-range:disabled::-webkit-slider-runnable-track{cursor:default}.custom-range:disabled::-moz-range-thumb{background-color:#adb5bd}.custom-range:disabled::-moz-range-track{cursor:default}.custom-range:disabled::-ms-thumb{background-color:#adb5bd}.custom-control-label:before,.custom-file-label,.custom-select{transition:background-color .15s ease-in-out,border-color .15s ease-in-out,box-shadow .15s ease-in-out}@media (prefers-reduced-motion:reduce){.custom-control-label:before,.custom-file-label,.custom-select{transition:none}}.nav{display:flex;flex-wrap:wrap;padding-left:0;margin-bottom:0;list-style:none}.nav-link{display:block;padding:.5rem 1rem}.nav-link:focus,.nav-link:hover{text-decoration:none}.nav-link.disabled{color:#6c757d;pointer-events:none;cursor:default}.nav-tabs{border-bottom:1px solid #dee2e6}.nav-tabs .nav-item{margin-bottom:-1px}.nav-tabs .nav-link{border:1px solid transparent;border-top-left-radius:.25rem;border-top-right-radius:.25rem}.nav-tabs .nav-link:focus,.nav-tabs .nav-link:hover{border-color:#e9ecef #e9ecef #dee2e6}.nav-tabs .nav-link.disabled{color:#6c757d;background-color:transparent;border-color:transparent}.nav-tabs .nav-item.show .nav-link,.nav-tabs .nav-link.active{color:#495057;background-color:#fff;border-color:#dee2e6 #dee2e6 #fff}.nav-tabs .dropdown-menu{margin-top:-1px;border-top-left-radius:0;border-top-right-radius:0}.nav-pills .nav-link{border-radius:.25rem}.nav-pills .nav-link.active,.nav-pills .show>.nav-link{color:#fff;background-color:#007bff}.nav-fill .nav-item{flex:1 1 auto;text-align:center}.nav-justified .nav-item{flex-basis:0;flex-grow:1;text-align:center}.tab-content>.tab-pane{display:none}.tab-content>.active{display:block}.navbar{position:relative;padding:.5rem 1rem}.navbar,.navbar .container,.navbar .container-fluid,.navbar .container-lg,.navbar .container-md,.navbar .container-sm,.navbar .container-xl{display:flex;flex-wrap:wrap;align-items:center;justify-content:space-between}.navbar-brand{display:inline-block;padding-top:.3125rem;padding-bottom:.3125rem;margin-right:1rem;font-size:1.25rem;line-height:inherit;white-space:nowrap}.navbar-brand:focus,.navbar-brand:hover{text-decoration:none}.navbar-nav{display:flex;flex-direction:column;padding-left:0;margin-bottom:0;list-style:none}.navbar-nav .nav-link{padding-right:0;padding-left:0}.navbar-nav .dropdown-menu{position:static;float:none}.navbar-text{display:inline-block;padding-top:.5rem;padding-bottom:.5rem}.navbar-collapse{flex-basis:100%;flex-grow:1;align-items:center}.navbar-toggler{padding:.25rem .75rem;font-size:1.25rem;line-height:1;background-color:transparent;border:1px solid transparent;border-radius:.25rem}.navbar-toggler:focus,.navbar-toggler:hover{text-decoration:none}.navbar-toggler-icon{display:inline-block;width:1.5em;height:1.5em;vertical-align:middle;content:"";background:no-repeat 50%;background-size:100% 100%}@media (max-width:575.98px){.navbar-expand-sm>.container,.navbar-expand-sm>.container-fluid,.navbar-expand-sm>.container-lg,.navbar-expand-sm>.container-md,.navbar-expand-sm>.container-sm,.navbar-expand-sm>.container-xl{padding-right:0;padding-left:0}}@media (min-width:576px){.navbar-expand-sm{flex-flow:row nowrap;justify-content:flex-start}.navbar-expand-sm .navbar-nav{flex-direction:row}.navbar-expand-sm .navbar-nav .dropdown-menu{position:absolute}.navbar-expand-sm .navbar-nav .nav-link{padding-right:.5rem;padding-left:.5rem}.navbar-expand-sm>.container,.navbar-expand-sm>.container-fluid,.navbar-expand-sm>.container-lg,.navbar-expand-sm>.container-md,.navbar-expand-sm>.container-sm,.navbar-expand-sm>.container-xl{flex-wrap:nowrap}.navbar-expand-sm .navbar-collapse{display:flex!important;flex-basis:auto}.navbar-expand-sm .navbar-toggler{display:none}}@media (max-width:767.98px){.navbar-expand-md>.container,.navbar-expand-md>.container-fluid,.navbar-expand-md>.container-lg,.navbar-expand-md>.container-md,.navbar-expand-md>.container-sm,.navbar-expand-md>.container-xl{padding-right:0;padding-left:0}}@media (min-width:768px){.navbar-expand-md{flex-flow:row nowrap;justify-content:flex-start}.navbar-expand-md .navbar-nav{flex-direction:row}.navbar-expand-md .navbar-nav .dropdown-menu{position:absolute}.navbar-expand-md .navbar-nav .nav-link{padding-right:.5rem;padding-left:.5rem}.navbar-expand-md>.container,.navbar-expand-md>.container-fluid,.navbar-expand-md>.container-lg,.navbar-expand-md>.container-md,.navbar-expand-md>.container-sm,.navbar-expand-md>.container-xl{flex-wrap:nowrap}.navbar-expand-md .navbar-collapse{display:flex!important;flex-basis:auto}.navbar-expand-md .navbar-toggler{display:none}}@media (max-width:991.98px){.navbar-expand-lg>.container,.navbar-expand-lg>.container-fluid,.navbar-expand-lg>.container-lg,.navbar-expand-lg>.container-md,.navbar-expand-lg>.container-sm,.navbar-expand-lg>.container-xl{padding-right:0;padding-left:0}}@media (min-width:992px){.navbar-expand-lg{flex-flow:row nowrap;justify-content:flex-start}.navbar-expand-lg .navbar-nav{flex-direction:row}.navbar-expand-lg .navbar-nav .dropdown-menu{position:absolute}.navbar-expand-lg .navbar-nav .nav-link{padding-right:.5rem;padding-left:.5rem}.navbar-expand-lg>.container,.navbar-expand-lg>.container-fluid,.navbar-expand-lg>.container-lg,.navbar-expand-lg>.container-md,.navbar-expand-lg>.container-sm,.navbar-expand-lg>.container-xl{flex-wrap:nowrap}.navbar-expand-lg .navbar-collapse{display:flex!important;flex-basis:auto}.navbar-expand-lg .navbar-toggler{display:none}}@media (max-width:1199.98px){.navbar-expand-xl>.container,.navbar-expand-xl>.container-fluid,.navbar-expand-xl>.container-lg,.navbar-expand-xl>.container-md,.navbar-expand-xl>.container-sm,.navbar-expand-xl>.container-xl{padding-right:0;padding-left:0}}@media (min-width:1200px){.navbar-expand-xl{flex-flow:row nowrap;justify-content:flex-start}.navbar-expand-xl .navbar-nav{flex-direction:row}.navbar-expand-xl .navbar-nav .dropdown-menu{position:absolute}.navbar-expand-xl .navbar-nav .nav-link{padding-right:.5rem;padding-left:.5rem}.navbar-expand-xl>.container,.navbar-expand-xl>.container-fluid,.navbar-expand-xl>.container-lg,.navbar-expand-xl>.container-md,.navbar-expand-xl>.container-sm,.navbar-expand-xl>.container-xl{flex-wrap:nowrap}.navbar-expand-xl .navbar-collapse{display:flex!important;flex-basis:auto}.navbar-expand-xl .navbar-toggler{display:none}}.navbar-expand{flex-flow:row nowrap;justify-content:flex-start}.navbar-expand>.container,.navbar-expand>.container-fluid,.navbar-expand>.container-lg,.navbar-expand>.container-md,.navbar-expand>.container-sm,.navbar-expand>.container-xl{padding-right:0;padding-left:0}.navbar-expand .navbar-nav{flex-direction:row}.navbar-expand .navbar-nav .dropdown-menu{position:absolute}.navbar-expand .navbar-nav .nav-link{padding-right:.5rem;padding-left:.5rem}.navbar-expand>.container,.navbar-expand>.container-fluid,.navbar-expand>.container-lg,.navbar-expand>.container-md,.navbar-expand>.container-sm,.navbar-expand>.container-xl{flex-wrap:nowrap}.navbar-expand .navbar-collapse{display:flex!important;flex-basis:auto}.navbar-expand .navbar-toggler{display:none}.navbar-light .navbar-brand,.navbar-light .navbar-brand:focus,.navbar-light .navbar-brand:hover{color:rgba(0,0,0,.9)}.navbar-light .navbar-nav .nav-link{color:rgba(0,0,0,.5)}.navbar-light .navbar-nav .nav-link:focus,.navbar-light .navbar-nav .nav-link:hover{color:rgba(0,0,0,.7)}.navbar-light .navbar-nav .nav-link.disabled{color:rgba(0,0,0,.3)}.navbar-light .navbar-nav .active>.nav-link,.navbar-light .navbar-nav .nav-link.active,.navbar-light .navbar-nav .nav-link.show,.navbar-light .navbar-nav .show>.nav-link{color:rgba(0,0,0,.9)}.navbar-light .navbar-toggler{color:rgba(0,0,0,.5);border-color:rgba(0,0,0,.1)}.navbar-light .navbar-toggler-icon{background-image:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='30' height='30'%3E%3Cpath stroke='rgba(0,0,0,0.5)' stroke-linecap='round' stroke-miterlimit='10' stroke-width='2' d='M4 7h22M4 15h22M4 23h22'/%3E%3C/svg%3E")}.navbar-light .navbar-text{color:rgba(0,0,0,.5)}.navbar-light .navbar-text a,.navbar-light .navbar-text a:focus,.navbar-light .navbar-text a:hover{color:rgba(0,0,0,.9)}.navbar-dark .navbar-brand,.navbar-dark .navbar-brand:focus,.navbar-dark .navbar-brand:hover{color:#fff}.navbar-dark .navbar-nav .nav-link{color:hsla(0,0%,100%,.5)}.navbar-dark .navbar-nav .nav-link:focus,.navbar-dark .navbar-nav .nav-link:hover{color:hsla(0,0%,100%,.75)}.navbar-dark .navbar-nav .nav-link.disabled{color:hsla(0,0%,100%,.25)}.navbar-dark .navbar-nav .active>.nav-link,.navbar-dark .navbar-nav .nav-link.active,.navbar-dark .navbar-nav .nav-link.show,.navbar-dark .navbar-nav .show>.nav-link{color:#fff}.navbar-dark .navbar-toggler{color:hsla(0,0%,100%,.5);border-color:hsla(0,0%,100%,.1)}.navbar-dark .navbar-toggler-icon{background-image:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' width='30' height='30'%3E%3Cpath stroke='rgba(255,255,255,0.5)' stroke-linecap='round' stroke-miterlimit='10' stroke-width='2' d='M4 7h22M4 15h22M4 23h22'/%3E%3C/svg%3E")}.navbar-dark .navbar-text{color:hsla(0,0%,100%,.5)}.navbar-dark .navbar-text a,.navbar-dark .navbar-text a:focus,.navbar-dark .navbar-text a:hover{color:#fff}.card{position:relative;display:flex;flex-direction:column;min-width:0;word-wrap:break-word;background-color:#fff;background-clip:border-box;border:1px solid rgba(0,0,0,.125);border-radius:.25rem}.card>hr{margin-right:0;margin-left:0}.card>.list-group{border-top:inherit;border-bottom:inherit}.card>.list-group:first-child{border-top-width:0;border-top-left-radius:calc(.25rem - 1px);border-top-right-radius:calc(.25rem - 1px)}.card>.list-group:last-child{border-bottom-width:0;border-bottom-right-radius:calc(.25rem - 1px);border-bottom-left-radius:calc(.25rem - 1px)}.card-body{flex:1 1 auto;min-height:1px;padding:1.25rem}.card-title{margin-bottom:.75rem}.card-subtitle{margin-top:-.375rem}.card-subtitle,.card-text:last-child{margin-bottom:0}.card-link:hover{text-decoration:none}.card-link+.card-link{margin-left:1.25rem}.card-header{padding:.75rem 1.25rem;margin-bottom:0;background-color:rgba(0,0,0,.03);border-bottom:1px solid rgba(0,0,0,.125)}.card-header:first-child{border-radius:calc(.25rem - 1px) calc(.25rem - 1px) 0 0}.card-header+.list-group .list-group-item:first-child{border-top:0}.card-footer{padding:.75rem 1.25rem;background-color:rgba(0,0,0,.03);border-top:1px solid rgba(0,0,0,.125)}.card-footer:last-child{border-radius:0 0 calc(.25rem - 1px) calc(.25rem - 1px)}.card-header-tabs{margin-bottom:-.75rem;border-bottom:0}.card-header-pills,.card-header-tabs{margin-right:-.625rem;margin-left:-.625rem}.card-img-overlay{position:absolute;top:0;right:0;bottom:0;left:0;padding:1.25rem}.card-img,.card-img-bottom,.card-img-top{flex-shrink:0;width:100%}.card-img,.card-img-top{border-top-left-radius:calc(.25rem - 1px);border-top-right-radius:calc(.25rem - 1px)}.card-img,.card-img-bottom{border-bottom-right-radius:calc(.25rem - 1px);border-bottom-left-radius:calc(.25rem - 1px)}.card-deck .card{margin-bottom:15px}@media (min-width:576px){.card-deck{display:flex;flex-flow:row wrap;margin-right:-15px;margin-left:-15px}.card-deck .card{flex:1 0 0%;margin-right:15px;margin-bottom:0;margin-left:15px}}.card-group>.card{margin-bottom:15px}@media (min-width:576px){.card-group{display:flex;flex-flow:row wrap}.card-group>.card{flex:1 0 0%;margin-bottom:0}.card-group>.card+.card{margin-left:0;border-left:0}.card-group>.card:not(:last-child){border-top-right-radius:0;border-bottom-right-radius:0}.card-group>.card:not(:last-child) .card-header,.card-group>.card:not(:last-child) .card-img-top{border-top-right-radius:0}.card-group>.card:not(:last-child) .card-footer,.card-group>.card:not(:last-child) .card-img-bottom{border-bottom-right-radius:0}.card-group>.card:not(:first-child){border-top-left-radius:0;border-bottom-left-radius:0}.card-group>.card:not(:first-child) .card-header,.card-group>.card:not(:first-child) .card-img-top{border-top-left-radius:0}.card-group>.card:not(:first-child) .card-footer,.card-group>.card:not(:first-child) .card-img-bottom{border-bottom-left-radius:0}}.card-columns .card{margin-bottom:.75rem}@media (min-width:576px){.card-columns{column-count:3;column-gap:1.25rem;orphans:1;widows:1}.card-columns .card{display:inline-block;width:100%}}.accordion>.card{overflow:hidden}.accordion>.card:not(:last-of-type){border-bottom:0;border-bottom-right-radius:0;border-bottom-left-radius:0}.accordion>.card:not(:first-of-type){border-top-left-radius:0;border-top-right-radius:0}.accordion>.card>.card-header{border-radius:0;margin-bottom:-1px}.breadcrumb{flex-wrap:wrap;padding:.75rem 1rem;margin-bottom:1rem;list-style:none;background-color:#e9ecef;border-radius:.25rem}.breadcrumb,.breadcrumb-item{display:flex}.breadcrumb-item+.breadcrumb-item{padding-left:.5rem}.breadcrumb-item+.breadcrumb-item:before{display:inline-block;padding-right:.5rem;color:#6c757d;content:"/"}.breadcrumb-item+.breadcrumb-item:hover:before{text-decoration:underline;text-decoration:none}.breadcrumb-item.active{color:#6c757d}.pagination{display:flex;padding-left:0;list-style:none;border-radius:.25rem}.page-link{position:relative;display:block;padding:.5rem .75rem;margin-left:-1px;line-height:1.25;color:#007bff;background-color:#fff;border:1px solid #dee2e6}.page-link:hover{z-index:2;color:#0056b3;text-decoration:none;background-color:#e9ecef;border-color:#dee2e6}.page-link:focus{z-index:3;outline:0;box-shadow:0 0 0 .2rem rgba(0,123,255,.25)}.page-item:first-child .page-link{margin-left:0;border-top-left-radius:.25rem;border-bottom-left-radius:.25rem}.page-item:last-child .page-link{border-top-right-radius:.25rem;border-bottom-right-radius:.25rem}.page-item.active .page-link{z-index:3;color:#fff;background-color:#007bff;border-color:#007bff}.page-item.disabled .page-link{color:#6c757d;pointer-events:none;cursor:auto;background-color:#fff;border-color:#dee2e6}.pagination-lg .page-link{padding:.75rem 1.5rem;font-size:1.25rem;line-height:1.5}.pagination-lg .page-item:first-child .page-link{border-top-left-radius:.3rem;border-bottom-left-radius:.3rem}.pagination-lg .page-item:last-child .page-link{border-top-right-radius:.3rem;border-bottom-right-radius:.3rem}.pagination-sm .page-link{padding:.25rem .5rem;font-size:.875rem;line-height:1.5}.pagination-sm .page-item:first-child .page-link{border-top-left-radius:.2rem;border-bottom-left-radius:.2rem}.pagination-sm .page-item:last-child .page-link{border-top-right-radius:.2rem;border-bottom-right-radius:.2rem}.badge{display:inline-block;padding:.25em .4em;font-size:75%;font-weight:700;line-height:1;text-align:center;white-space:nowrap;vertical-align:baseline;border-radius:.25rem;transition:color .15s ease-in-out,background-color .15s ease-in-out,border-color .15s ease-in-out,box-shadow .15s ease-in-out}@media (prefers-reduced-motion:reduce){.badge{transition:none}}a.badge:focus,a.badge:hover{text-decoration:none}.badge:empty{display:none}.btn .badge{position:relative;top:-1px}.badge-pill{padding-right:.6em;padding-left:.6em;border-radius:10rem}.badge-primary{color:#fff;background-color:#007bff}a.badge-primary:focus,a.badge-primary:hover{color:#fff;background-color:#0062cc}a.badge-primary.focus,a.badge-primary:focus{outline:0;box-shadow:0 0 0 .2rem rgba(0,123,255,.5)}.badge-secondary{color:#fff;background-color:#6c757d}a.badge-secondary:focus,a.badge-secondary:hover{color:#fff;background-color:#545b62}a.badge-secondary.focus,a.badge-secondary:focus{outline:0;box-shadow:0 0 0 .2rem rgba(108,117,125,.5)}.badge-success{color:#fff;background-color:#28a745}a.badge-success:focus,a.badge-success:hover{color:#fff;background-color:#1e7e34}a.badge-success.focus,a.badge-success:focus{outline:0;box-shadow:0 0 0 .2rem rgba(40,167,69,.5)}.badge-info{color:#fff;background-color:#17a2b8}a.badge-info:focus,a.badge-info:hover{color:#fff;background-color:#117a8b}a.badge-info.focus,a.badge-info:focus{outline:0;box-shadow:0 0 0 .2rem rgba(23,162,184,.5)}.badge-warning{color:#212529;background-color:#ffc107}a.badge-warning:focus,a.badge-warning:hover{color:#212529;background-color:#d39e00}a.badge-warning.focus,a.badge-warning:focus{outline:0;box-shadow:0 0 0 .2rem rgba(255,193,7,.5)}.badge-danger{color:#fff;background-color:#dc3545}a.badge-danger:focus,a.badge-danger:hover{color:#fff;background-color:#bd2130}a.badge-danger.focus,a.badge-danger:focus{outline:0;box-shadow:0 0 0 .2rem rgba(220,53,69,.5)}.badge-light{color:#212529;background-color:#f8f9fa}a.badge-light:focus,a.badge-light:hover{color:#212529;background-color:#dae0e5}a.badge-light.focus,a.badge-light:focus{outline:0;box-shadow:0 0 0 .2rem rgba(248,249,250,.5)}.badge-dark{color:#fff;background-color:#343a40}a.badge-dark:focus,a.badge-dark:hover{color:#fff;background-color:#1d2124}a.badge-dark.focus,a.badge-dark:focus{outline:0;box-shadow:0 0 0 .2rem rgba(52,58,64,.5)}.jumbotron{padding:2rem 1rem;margin-bottom:2rem;background-color:#e9ecef;border-radius:.3rem}@media (min-width:576px){.jumbotron{padding:4rem 2rem}}.jumbotron-fluid{padding-right:0;padding-left:0;border-radius:0}.alert{position:relative;padding:.75rem 1.25rem;margin-bottom:1rem;border:1px solid transparent;border-radius:.25rem}.alert-heading{color:inherit}.alert-link{font-weight:700}.alert-dismissible{padding-right:4rem}.alert-dismissible .close{position:absolute;top:0;right:0;padding:.75rem 1.25rem;color:inherit}.alert-primary{color:#004085;background-color:#cce5ff;border-color:#b8daff}.alert-primary hr{border-top-color:#9fcdff}.alert-primary .alert-link{color:#002752}.alert-secondary{color:#383d41;background-color:#e2e3e5;border-color:#d6d8db}.alert-secondary hr{border-top-color:#c8cbcf}.alert-secondary .alert-link{color:#202326}.alert-success{color:#155724;background-color:#d4edda;border-color:#c3e6cb}.alert-success hr{border-top-color:#b1dfbb}.alert-success .alert-link{color:#0b2e13}.alert-info{color:#0c5460;background-color:#d1ecf1;border-color:#bee5eb}.alert-info hr{border-top-color:#abdde5}.alert-info .alert-link{color:#062c33}.alert-warning{color:#856404;background-color:#fff3cd;border-color:#ffeeba}.alert-warning hr{border-top-color:#ffe8a1}.alert-warning .alert-link{color:#533f03}.alert-danger{color:#721c24;background-color:#f8d7da;border-color:#f5c6cb}.alert-danger hr{border-top-color:#f1b0b7}.alert-danger .alert-link{color:#491217}.alert-light{color:#818182;background-color:#fefefe;border-color:#fdfdfe}.alert-light hr{border-top-color:#ececf6}.alert-light .alert-link{color:#686868}.alert-dark{color:#1b1e21;background-color:#d6d8d9;border-color:#c6c8ca}.alert-dark hr{border-top-color:#b9bbbe}.alert-dark .alert-link{color:#040505}@keyframes progress-bar-stripes{0%{background-position:1rem 0}to{background-position:0 0}}.progress{height:1rem;line-height:0;font-size:.75rem;background-color:#e9ecef;border-radius:.25rem}.progress,.progress-bar{display:flex;overflow:hidden}.progress-bar{flex-direction:column;justify-content:center;color:#fff;text-align:center;white-space:nowrap;background-color:#007bff;transition:width .6s ease}@media (prefers-reduced-motion:reduce){.progress-bar{transition:none}}.progress-bar-striped{background-image:linear-gradient(45deg,hsla(0,0%,100%,.15) 25%,transparent 0,transparent 50%,hsla(0,0%,100%,.15) 0,hsla(0,0%,100%,.15) 75%,transparent 0,transparent);background-size:1rem 1rem}.progress-bar-animated{animation:progress-bar-stripes 1s linear infinite}@media (prefers-reduced-motion:reduce){.progress-bar-animated{animation:none}}.media{display:flex;align-items:flex-start}.media-body{flex:1}.list-group{display:flex;flex-direction:column;padding-left:0;margin-bottom:0;border-radius:.25rem}.list-group-item-action{width:100%;color:#495057;text-align:inherit}.list-group-item-action:focus,.list-group-item-action:hover{z-index:1;color:#495057;text-decoration:none;background-color:#f8f9fa}.list-group-item-action:active{color:#212529;background-color:#e9ecef}.list-group-item{position:relative;display:block;padding:.75rem 1.25rem;background-color:#fff;border:1px solid rgba(0,0,0,.125)}.list-group-item:first-child{border-top-left-radius:inherit;border-top-right-radius:inherit}.list-group-item:last-child{border-bottom-right-radius:inherit;border-bottom-left-radius:inherit}.list-group-item.disabled,.list-group-item:disabled{color:#6c757d;pointer-events:none;background-color:#fff}.list-group-item.active{z-index:2;color:#fff;background-color:#007bff;border-color:#007bff}.list-group-item+.list-group-item{border-top-width:0}.list-group-item+.list-group-item.active{margin-top:-1px;border-top-width:1px}.list-group-horizontal{flex-direction:row}.list-group-horizontal>.list-group-item:first-child{border-bottom-left-radius:.25rem;border-top-right-radius:0}.list-group-horizontal>.list-group-item:last-child{border-top-right-radius:.25rem;border-bottom-left-radius:0}.list-group-horizontal>.list-group-item.active{margin-top:0}.list-group-horizontal>.list-group-item+.list-group-item{border-top-width:1px;border-left-width:0}.list-group-horizontal>.list-group-item+.list-group-item.active{margin-left:-1px;border-left-width:1px}@media (min-width:576px){.list-group-horizontal-sm{flex-direction:row}.list-group-horizontal-sm>.list-group-item:first-child{border-bottom-left-radius:.25rem;border-top-right-radius:0}.list-group-horizontal-sm>.list-group-item:last-child{border-top-right-radius:.25rem;border-bottom-left-radius:0}.list-group-horizontal-sm>.list-group-item.active{margin-top:0}.list-group-horizontal-sm>.list-group-item+.list-group-item{border-top-width:1px;border-left-width:0}.list-group-horizontal-sm>.list-group-item+.list-group-item.active{margin-left:-1px;border-left-width:1px}}@media (min-width:768px){.list-group-horizontal-md{flex-direction:row}.list-group-horizontal-md>.list-group-item:first-child{border-bottom-left-radius:.25rem;border-top-right-radius:0}.list-group-horizontal-md>.list-group-item:last-child{border-top-right-radius:.25rem;border-bottom-left-radius:0}.list-group-horizontal-md>.list-group-item.active{margin-top:0}.list-group-horizontal-md>.list-group-item+.list-group-item{border-top-width:1px;border-left-width:0}.list-group-horizontal-md>.list-group-item+.list-group-item.active{margin-left:-1px;border-left-width:1px}}@media (min-width:992px){.list-group-horizontal-lg{flex-direction:row}.list-group-horizontal-lg>.list-group-item:first-child{border-bottom-left-radius:.25rem;border-top-right-radius:0}.list-group-horizontal-lg>.list-group-item:last-child{border-top-right-radius:.25rem;border-bottom-left-radius:0}.list-group-horizontal-lg>.list-group-item.active{margin-top:0}.list-group-horizontal-lg>.list-group-item+.list-group-item{border-top-width:1px;border-left-width:0}.list-group-horizontal-lg>.list-group-item+.list-group-item.active{margin-left:-1px;border-left-width:1px}}@media (min-width:1200px){.list-group-horizontal-xl{flex-direction:row}.list-group-horizontal-xl>.list-group-item:first-child{border-bottom-left-radius:.25rem;border-top-right-radius:0}.list-group-horizontal-xl>.list-group-item:last-child{border-top-right-radius:.25rem;border-bottom-left-radius:0}.list-group-horizontal-xl>.list-group-item.active{margin-top:0}.list-group-horizontal-xl>.list-group-item+.list-group-item{border-top-width:1px;border-left-width:0}.list-group-horizontal-xl>.list-group-item+.list-group-item.active{margin-left:-1px;border-left-width:1px}}.list-group-flush{border-radius:0}.list-group-flush>.list-group-item{border-width:0 0 1px}.list-group-flush>.list-group-item:last-child{border-bottom-width:0}.list-group-item-primary{color:#004085;background-color:#b8daff}.list-group-item-primary.list-group-item-action:focus,.list-group-item-primary.list-group-item-action:hover{color:#004085;background-color:#9fcdff}.list-group-item-primary.list-group-item-action.active{color:#fff;background-color:#004085;border-color:#004085}.list-group-item-secondary{color:#383d41;background-color:#d6d8db}.list-group-item-secondary.list-group-item-action:focus,.list-group-item-secondary.list-group-item-action:hover{color:#383d41;background-color:#c8cbcf}.list-group-item-secondary.list-group-item-action.active{color:#fff;background-color:#383d41;border-color:#383d41}.list-group-item-success{color:#155724;background-color:#c3e6cb}.list-group-item-success.list-group-item-action:focus,.list-group-item-success.list-group-item-action:hover{color:#155724;background-color:#b1dfbb}.list-group-item-success.list-group-item-action.active{color:#fff;background-color:#155724;border-color:#155724}.list-group-item-info{color:#0c5460;background-color:#bee5eb}.list-group-item-info.list-group-item-action:focus,.list-group-item-info.list-group-item-action:hover{color:#0c5460;background-color:#abdde5}.list-group-item-info.list-group-item-action.active{color:#fff;background-color:#0c5460;border-color:#0c5460}.list-group-item-warning{color:#856404;background-color:#ffeeba}.list-group-item-warning.list-group-item-action:focus,.list-group-item-warning.list-group-item-action:hover{color:#856404;background-color:#ffe8a1}.list-group-item-warning.list-group-item-action.active{color:#fff;background-color:#856404;border-color:#856404}.list-group-item-danger{color:#721c24;background-color:#f5c6cb}.list-group-item-danger.list-group-item-action:focus,.list-group-item-danger.list-group-item-action:hover{color:#721c24;background-color:#f1b0b7}.list-group-item-danger.list-group-item-action.active{color:#fff;background-color:#721c24;border-color:#721c24}.list-group-item-light{color:#818182;background-color:#fdfdfe}.list-group-item-light.list-group-item-action:focus,.list-group-item-light.list-group-item-action:hover{color:#818182;background-color:#ececf6}.list-group-item-light.list-group-item-action.active{color:#fff;background-color:#818182;border-color:#818182}.list-group-item-dark{color:#1b1e21;background-color:#c6c8ca}.list-group-item-dark.list-group-item-action:focus,.list-group-item-dark.list-group-item-action:hover{color:#1b1e21;background-color:#b9bbbe}.list-group-item-dark.list-group-item-action.active{color:#fff;background-color:#1b1e21;border-color:#1b1e21}.close{float:right;font-size:1.5rem;font-weight:700;line-height:1;color:#000;text-shadow:0 1px 0 #fff;opacity:.5}.close:hover{color:#000;text-decoration:none}.close:not(:disabled):not(.disabled):focus,.close:not(:disabled):not(.disabled):hover{opacity:.75}button.close{padding:0;background-color:transparent;border:0}a.close.disabled{pointer-events:none}.toast{max-width:350px;overflow:hidden;font-size:.875rem;background-color:hsla(0,0%,100%,.85);background-clip:padding-box;border:1px solid rgba(0,0,0,.1);box-shadow:0 .25rem .75rem rgba(0,0,0,.1);backdrop-filter:blur(10px);opacity:0;border-radius:.25rem}.toast:not(:last-child){margin-bottom:.75rem}.toast.showing{opacity:1}.toast.show{display:block;opacity:1}.toast.hide{display:none}.toast-header{display:flex;align-items:center;padding:.25rem .75rem;color:#6c757d;background-color:hsla(0,0%,100%,.85);background-clip:padding-box;border-bottom:1px solid rgba(0,0,0,.05)}.toast-body{padding:.75rem}.modal-open{overflow:hidden}.modal-open .modal{overflow-x:hidden;overflow-y:auto}.modal{position:fixed;top:0;left:0;z-index:1050;display:none;width:100%;height:100%;overflow:hidden;outline:0}.modal-dialog{position:relative;width:auto;margin:.5rem;pointer-events:none}.modal.fade .modal-dialog{transition:transform .3s ease-out;transform:translateY(-50px)}@media (prefers-reduced-motion:reduce){.modal.fade .modal-dialog{transition:none}}.modal.show .modal-dialog{transform:none}.modal.modal-static .modal-dialog{transform:scale(1.02)}.modal-dialog-scrollable{display:flex;max-height:calc(100% - 1rem)}.modal-dialog-scrollable .modal-content{max-height:calc(100vh - 1rem);overflow:hidden}.modal-dialog-scrollable .modal-footer,.modal-dialog-scrollable .modal-header{flex-shrink:0}.modal-dialog-scrollable .modal-body{overflow-y:auto}.modal-dialog-centered{display:flex;align-items:center;min-height:calc(100% - 1rem)}.modal-dialog-centered:before{display:block;height:calc(100vh - 1rem);height:min-content;content:""}.modal-dialog-centered.modal-dialog-scrollable{flex-direction:column;justify-content:center;height:100%}.modal-dialog-centered.modal-dialog-scrollable .modal-content{max-height:none}.modal-dialog-centered.modal-dialog-scrollable:before{content:none}.modal-content{position:relative;display:flex;flex-direction:column;width:100%;pointer-events:auto;background-color:#fff;background-clip:padding-box;border:1px solid rgba(0,0,0,.2);border-radius:.3rem;outline:0}.modal-backdrop{position:fixed;top:0;left:0;z-index:1040;width:100vw;height:100vh;background-color:#000}.modal-backdrop.fade{opacity:0}.modal-backdrop.show{opacity:.5}.modal-header{display:flex;align-items:flex-start;justify-content:space-between;padding:1rem;border-bottom:1px solid #dee2e6;border-top-left-radius:calc(.3rem - 1px);border-top-right-radius:calc(.3rem - 1px)}.modal-header .close{padding:1rem;margin:-1rem -1rem -1rem auto}.modal-title{margin-bottom:0;line-height:1.5}.modal-body{position:relative;flex:1 1 auto;padding:1rem}.modal-footer{display:flex;flex-wrap:wrap;align-items:center;justify-content:flex-end;padding:.75rem;border-top:1px solid #dee2e6;border-bottom-right-radius:calc(.3rem - 1px);border-bottom-left-radius:calc(.3rem - 1px)}.modal-footer>*{margin:.25rem}.modal-scrollbar-measure{position:absolute;top:-9999px;width:50px;height:50px;overflow:scroll}@media (min-width:576px){.modal-dialog{max-width:500px;margin:1.75rem auto}.modal-dialog-scrollable{max-height:calc(100% - 3.5rem)}.modal-dialog-scrollable .modal-content{max-height:calc(100vh - 3.5rem)}.modal-dialog-centered{min-height:calc(100% - 3.5rem)}.modal-dialog-centered:before{height:calc(100vh - 3.5rem);height:min-content}.modal-sm{max-width:300px}}@media (min-width:992px){.modal-lg,.modal-xl{max-width:800px}}@media (min-width:1200px){.modal-xl{max-width:1140px}}.tooltip{position:absolute;z-index:1070;display:block;margin:0;font-family:-apple-system,BlinkMacSystemFont,Segoe UI,Roboto,Helvetica Neue,Arial,Noto Sans,sans-serif,Apple Color Emoji,Segoe UI Emoji,Segoe UI Symbol,Noto Color Emoji;font-style:normal;font-weight:400;line-height:1.5;text-align:left;text-align:start;text-decoration:none;text-shadow:none;text-transform:none;letter-spacing:normal;word-break:normal;word-spacing:normal;white-space:normal;line-break:auto;font-size:.875rem;word-wrap:break-word;opacity:0}.tooltip.show{opacity:.9}.tooltip .arrow{position:absolute;display:block;width:.8rem;height:.4rem}.tooltip .arrow:before{position:absolute;content:"";border-color:transparent;border-style:solid}.bs-tooltip-auto[x-placement^=top],.bs-tooltip-top{padding:.4rem 0}.bs-tooltip-auto[x-placement^=top] .arrow,.bs-tooltip-top .arrow{bottom:0}.bs-tooltip-auto[x-placement^=top] .arrow:before,.bs-tooltip-top .arrow:before{top:0;border-width:.4rem .4rem 0;border-top-color:#000}.bs-tooltip-auto[x-placement^=right],.bs-tooltip-right{padding:0 .4rem}.bs-tooltip-auto[x-placement^=right] .arrow,.bs-tooltip-right .arrow{left:0;width:.4rem;height:.8rem}.bs-tooltip-auto[x-placement^=right] .arrow:before,.bs-tooltip-right .arrow:before{right:0;border-width:.4rem .4rem .4rem 0;border-right-color:#000}.bs-tooltip-auto[x-placement^=bottom],.bs-tooltip-bottom{padding:.4rem 0}.bs-tooltip-auto[x-placement^=bottom] .arrow,.bs-tooltip-bottom .arrow{top:0}.bs-tooltip-auto[x-placement^=bottom] .arrow:before,.bs-tooltip-bottom .arrow:before{bottom:0;border-width:0 .4rem .4rem;border-bottom-color:#000}.bs-tooltip-auto[x-placement^=left],.bs-tooltip-left{padding:0 .4rem}.bs-tooltip-auto[x-placement^=left] .arrow,.bs-tooltip-left .arrow{right:0;width:.4rem;height:.8rem}.bs-tooltip-auto[x-placement^=left] .arrow:before,.bs-tooltip-left .arrow:before{left:0;border-width:.4rem 0 .4rem .4rem;border-left-color:#000}.tooltip-inner{max-width:200px;padding:.25rem .5rem;color:#fff;text-align:center;background-color:#000;border-radius:.25rem}.popover{top:0;left:0;z-index:1060;max-width:276px;font-family:-apple-system,BlinkMacSystemFont,Segoe UI,Roboto,Helvetica Neue,Arial,Noto Sans,sans-serif,Apple Color Emoji,Segoe UI Emoji,Segoe UI Symbol,Noto Color Emoji;font-style:normal;font-weight:400;line-height:1.5;text-align:left;text-align:start;text-decoration:none;text-shadow:none;text-transform:none;letter-spacing:normal;word-break:normal;word-spacing:normal;white-space:normal;line-break:auto;font-size:.875rem;word-wrap:break-word;background-color:#fff;background-clip:padding-box;border:1px solid rgba(0,0,0,.2);border-radius:.3rem}.popover,.popover .arrow{position:absolute;display:block}.popover .arrow{width:1rem;height:.5rem;margin:0 .3rem}.popover .arrow:after,.popover .arrow:before{position:absolute;display:block;content:"";border-color:transparent;border-style:solid}.bs-popover-auto[x-placement^=top],.bs-popover-top{margin-bottom:.5rem}.bs-popover-auto[x-placement^=top]>.arrow,.bs-popover-top>.arrow{bottom:calc(-.5rem - 1px)}.bs-popover-auto[x-placement^=top]>.arrow:before,.bs-popover-top>.arrow:before{bottom:0;border-width:.5rem .5rem 0;border-top-color:rgba(0,0,0,.25)}.bs-popover-auto[x-placement^=top]>.arrow:after,.bs-popover-top>.arrow:after{bottom:1px;border-width:.5rem .5rem 0;border-top-color:#fff}.bs-popover-auto[x-placement^=right],.bs-popover-right{margin-left:.5rem}.bs-popover-auto[x-placement^=right]>.arrow,.bs-popover-right>.arrow{left:calc(-.5rem - 1px);width:.5rem;height:1rem;margin:.3rem 0}.bs-popover-auto[x-placement^=right]>.arrow:before,.bs-popover-right>.arrow:before{left:0;border-width:.5rem .5rem .5rem 0;border-right-color:rgba(0,0,0,.25)}.bs-popover-auto[x-placement^=right]>.arrow:after,.bs-popover-right>.arrow:after{left:1px;border-width:.5rem .5rem .5rem 0;border-right-color:#fff}.bs-popover-auto[x-placement^=bottom],.bs-popover-bottom{margin-top:.5rem}.bs-popover-auto[x-placement^=bottom]>.arrow,.bs-popover-bottom>.arrow{top:calc(-.5rem - 1px)}.bs-popover-auto[x-placement^=bottom]>.arrow:before,.bs-popover-bottom>.arrow:before{top:0;border-width:0 .5rem .5rem;border-bottom-color:rgba(0,0,0,.25)}.bs-popover-auto[x-placement^=bottom]>.arrow:after,.bs-popover-bottom>.arrow:after{top:1px;border-width:0 .5rem .5rem;border-bottom-color:#fff}.bs-popover-auto[x-placement^=bottom] .popover-header:before,.bs-popover-bottom .popover-header:before{position:absolute;top:0;left:50%;display:block;width:1rem;margin-left:-.5rem;content:"";border-bottom:1px solid #f7f7f7}.bs-popover-auto[x-placement^=left],.bs-popover-left{margin-right:.5rem}.bs-popover-auto[x-placement^=left]>.arrow,.bs-popover-left>.arrow{right:calc(-.5rem - 1px);width:.5rem;height:1rem;margin:.3rem 0}.bs-popover-auto[x-placement^=left]>.arrow:before,.bs-popover-left>.arrow:before{right:0;border-width:.5rem 0 .5rem .5rem;border-left-color:rgba(0,0,0,.25)}.bs-popover-auto[x-placement^=left]>.arrow:after,.bs-popover-left>.arrow:after{right:1px;border-width:.5rem 0 .5rem .5rem;border-left-color:#fff}.popover-header{padding:.5rem .75rem;margin-bottom:0;font-size:1rem;background-color:#f7f7f7;border-bottom:1px solid #ebebeb;border-top-left-radius:calc(.3rem - 1px);border-top-right-radius:calc(.3rem - 1px)}.popover-header:empty{display:none}.popover-body{padding:.5rem .75rem;color:#212529}.carousel{position:relative}.carousel.pointer-event{touch-action:pan-y}.carousel-inner{position:relative;width:100%;overflow:hidden}.carousel-inner:after{display:block;clear:both;content:""}.carousel-item{position:relative;display:none;float:left;width:100%;margin-right:-100%;backface-visibility:hidden;transition:transform .6s ease-in-out}@media (prefers-reduced-motion:reduce){.carousel-item{transition:none}}.carousel-item-next,.carousel-item-prev,.carousel-item.active{display:block}.active.carousel-item-right,.carousel-item-next:not(.carousel-item-left){transform:translateX(100%)}.active.carousel-item-left,.carousel-item-prev:not(.carousel-item-right){transform:translateX(-100%)}.carousel-fade .carousel-item{opacity:0;transition-property:opacity;transform:none}.carousel-fade .carousel-item-next.carousel-item-left,.carousel-fade .carousel-item-prev.carousel-item-right,.carousel-fade .carousel-item.active{z-index:1;opacity:1}.carousel-fade .active.carousel-item-left,.carousel-fade .active.carousel-item-right{z-index:0;opacity:0;transition:opacity 0s .6s}@media (prefers-reduced-motion:reduce){.carousel-fade .active.carousel-item-left,.carousel-fade .active.carousel-item-right{transition:none}}.carousel-control-next,.carousel-control-prev{position:absolute;top:0;bottom:0;z-index:1;display:flex;align-items:center;justify-content:center;width:15%;color:#fff;text-align:center;opacity:.5;transition:opacity .15s ease}@media (prefers-reduced-motion:reduce){.carousel-control-next,.carousel-control-prev{transition:none}}.carousel-control-next:focus,.carousel-control-next:hover,.carousel-control-prev:focus,.carousel-control-prev:hover{color:#fff;text-decoration:none;outline:0;opacity:.9}.carousel-control-prev{left:0}.carousel-control-next{right:0}.carousel-control-next-icon,.carousel-control-prev-icon{display:inline-block;width:20px;height:20px;background:no-repeat 50%/100% 100%}.carousel-control-prev-icon{background-image:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' fill='%23fff' width='8' height='8'%3E%3Cpath d='M5.25 0l-4 4 4 4 1.5-1.5L4.25 4l2.5-2.5L5.25 0z'/%3E%3C/svg%3E")}.carousel-control-next-icon{background-image:url("data:image/svg+xml;charset=utf-8,%3Csvg xmlns='http://www.w3.org/2000/svg' fill='%23fff' width='8' height='8'%3E%3Cpath d='M2.75 0l-1.5 1.5L3.75 4l-2.5 2.5L2.75 8l4-4-4-4z'/%3E%3C/svg%3E")}.carousel-indicators{position:absolute;right:0;bottom:0;left:0;z-index:15;display:flex;justify-content:center;padding-left:0;margin-right:15%;margin-left:15%;list-style:none}.carousel-indicators li{box-sizing:content-box;flex:0 1 auto;width:30px;height:3px;margin-right:3px;margin-left:3px;text-indent:-999px;cursor:pointer;background-color:#fff;background-clip:padding-box;border-top:10px solid transparent;border-bottom:10px solid transparent;opacity:.5;transition:opacity .6s ease}@media (prefers-reduced-motion:reduce){.carousel-indicators li{transition:none}}.carousel-indicators .active{opacity:1}.carousel-caption{position:absolute;right:15%;bottom:20px;left:15%;z-index:10;padding-top:20px;padding-bottom:20px;color:#fff;text-align:center}@keyframes spinner-border{to{transform:rotate(1turn)}}.spinner-border{display:inline-block;width:2rem;height:2rem;vertical-align:text-bottom;border:.25em solid;border-right:.25em solid transparent;border-radius:50%;animation:spinner-border .75s linear infinite}.spinner-border-sm{width:1rem;height:1rem;border-width:.2em}@keyframes spinner-grow{0%{transform:scale(0)}50%{opacity:1;transform:none}}.spinner-grow{display:inline-block;width:2rem;height:2rem;vertical-align:text-bottom;background-color:currentColor;border-radius:50%;opacity:0;animation:spinner-grow .75s linear infinite}.spinner-grow-sm{width:1rem;height:1rem}.align-baseline{vertical-align:baseline!important}.align-top{vertical-align:top!important}.align-middle{vertical-align:middle!important}.align-bottom{vertical-align:bottom!important}.align-text-bottom{vertical-align:text-bottom!important}.align-text-top{vertical-align:text-top!important}.bg-primary{background-color:#007bff!important}a.bg-primary:focus,a.bg-primary:hover,button.bg-primary:focus,button.bg-primary:hover{background-color:#0062cc!important}.bg-secondary{background-color:#6c757d!important}a.bg-secondary:focus,a.bg-secondary:hover,button.bg-secondary:focus,button.bg-secondary:hover{background-color:#545b62!important}.bg-success{background-color:#28a745!important}a.bg-success:focus,a.bg-success:hover,button.bg-success:focus,button.bg-success:hover{background-color:#1e7e34!important}.bg-info{background-color:#17a2b8!important}a.bg-info:focus,a.bg-info:hover,button.bg-info:focus,button.bg-info:hover{background-color:#117a8b!important}.bg-warning{background-color:#ffc107!important}a.bg-warning:focus,a.bg-warning:hover,button.bg-warning:focus,button.bg-warning:hover{background-color:#d39e00!important}.bg-danger{background-color:#dc3545!important}a.bg-danger:focus,a.bg-danger:hover,button.bg-danger:focus,button.bg-danger:hover{background-color:#bd2130!important}.bg-light{background-color:#f8f9fa!important}a.bg-light:focus,a.bg-light:hover,button.bg-light:focus,button.bg-light:hover{background-color:#dae0e5!important}.bg-dark{background-color:#343a40!important}a.bg-dark:focus,a.bg-dark:hover,button.bg-dark:focus,button.bg-dark:hover{background-color:#1d2124!important}.bg-white{background-color:#fff!important}.bg-transparent{background-color:transparent!important}.border{border:1px solid #dee2e6!important}.border-top{border-top:1px solid #dee2e6!important}.border-right{border-right:1px solid #dee2e6!important}.border-bottom{border-bottom:1px solid #dee2e6!important}.border-left{border-left:1px solid #dee2e6!important}.border-0{border:0!important}.border-top-0{border-top:0!important}.border-right-0{border-right:0!important}.border-bottom-0{border-bottom:0!important}.border-left-0{border-left:0!important}.border-primary{border-color:#007bff!important}.border-secondary{border-color:#6c757d!important}.border-success{border-color:#28a745!important}.border-info{border-color:#17a2b8!important}.border-warning{border-color:#ffc107!important}.border-danger{border-color:#dc3545!important}.border-light{border-color:#f8f9fa!important}.border-dark{border-color:#343a40!important}.border-white{border-color:#fff!important}.rounded-sm{border-radius:.2rem!important}.rounded{border-radius:.25rem!important}.rounded-top{border-top-left-radius:.25rem!important}.rounded-right,.rounded-top{border-top-right-radius:.25rem!important}.rounded-bottom,.rounded-right{border-bottom-right-radius:.25rem!important}.rounded-bottom,.rounded-left{border-bottom-left-radius:.25rem!important}.rounded-left{border-top-left-radius:.25rem!important}.rounded-lg{border-radius:.3rem!important}.rounded-circle{border-radius:50%!important}.rounded-pill{border-radius:50rem!important}.rounded-0{border-radius:0!important}.clearfix:after{display:block;clear:both;content:""}.d-none{display:none!important}.d-inline{display:inline!important}.d-inline-block{display:inline-block!important}.d-block{display:block!important}.d-table{display:table!important}.d-table-row{display:table-row!important}.d-table-cell{display:table-cell!important}.d-flex{display:flex!important}.d-inline-flex{display:inline-flex!important}@media (min-width:576px){.d-sm-none{display:none!important}.d-sm-inline{display:inline!important}.d-sm-inline-block{display:inline-block!important}.d-sm-block{display:block!important}.d-sm-table{display:table!important}.d-sm-table-row{display:table-row!important}.d-sm-table-cell{display:table-cell!important}.d-sm-flex{display:flex!important}.d-sm-inline-flex{display:inline-flex!important}}@media (min-width:768px){.d-md-none{display:none!important}.d-md-inline{display:inline!important}.d-md-inline-block{display:inline-block!important}.d-md-block{display:block!important}.d-md-table{display:table!important}.d-md-table-row{display:table-row!important}.d-md-table-cell{display:table-cell!important}.d-md-flex{display:flex!important}.d-md-inline-flex{display:inline-flex!important}}@media (min-width:992px){.d-lg-none{display:none!important}.d-lg-inline{display:inline!important}.d-lg-inline-block{display:inline-block!important}.d-lg-block{display:block!important}.d-lg-table{display:table!important}.d-lg-table-row{display:table-row!important}.d-lg-table-cell{display:table-cell!important}.d-lg-flex{display:flex!important}.d-lg-inline-flex{display:inline-flex!important}}@media (min-width:1200px){.d-xl-none{display:none!important}.d-xl-inline{display:inline!important}.d-xl-inline-block{display:inline-block!important}.d-xl-block{display:block!important}.d-xl-table{display:table!important}.d-xl-table-row{display:table-row!important}.d-xl-table-cell{display:table-cell!important}.d-xl-flex{display:flex!important}.d-xl-inline-flex{display:inline-flex!important}}@media print{.d-print-none{display:none!important}.d-print-inline{display:inline!important}.d-print-inline-block{display:inline-block!important}.d-print-block{display:block!important}.d-print-table{display:table!important}.d-print-table-row{display:table-row!important}.d-print-table-cell{display:table-cell!important}.d-print-flex{display:flex!important}.d-print-inline-flex{display:inline-flex!important}}.embed-responsive{position:relative;display:block;width:100%;padding:0;overflow:hidden}.embed-responsive:before{display:block;content:""}.embed-responsive .embed-responsive-item,.embed-responsive embed,.embed-responsive iframe,.embed-responsive object,.embed-responsive video{position:absolute;top:0;bottom:0;left:0;width:100%;height:100%;border:0}.embed-responsive-21by9:before{padding-top:42.85714%}.embed-responsive-16by9:before{padding-top:56.25%}.embed-responsive-4by3:before{padding-top:75%}.embed-responsive-1by1:before{padding-top:100%}.flex-row{flex-direction:row!important}.flex-column{flex-direction:column!important}.flex-row-reverse{flex-direction:row-reverse!important}.flex-column-reverse{flex-direction:column-reverse!important}.flex-wrap{flex-wrap:wrap!important}.flex-nowrap{flex-wrap:nowrap!important}.flex-wrap-reverse{flex-wrap:wrap-reverse!important}.flex-fill{flex:1 1 auto!important}.flex-grow-0{flex-grow:0!important}.flex-grow-1{flex-grow:1!important}.flex-shrink-0{flex-shrink:0!important}.flex-shrink-1{flex-shrink:1!important}.justify-content-start{justify-content:flex-start!important}.justify-content-end{justify-content:flex-end!important}.justify-content-center{justify-content:center!important}.justify-content-between{justify-content:space-between!important}.justify-content-around{justify-content:space-around!important}.align-items-start{align-items:flex-start!important}.align-items-end{align-items:flex-end!important}.align-items-center{align-items:center!important}.align-items-baseline{align-items:baseline!important}.align-items-stretch{align-items:stretch!important}.align-content-start{align-content:flex-start!important}.align-content-end{align-content:flex-end!important}.align-content-center{align-content:center!important}.align-content-between{align-content:space-between!important}.align-content-around{align-content:space-around!important}.align-content-stretch{align-content:stretch!important}.align-self-auto{align-self:auto!important}.align-self-start{align-self:flex-start!important}.align-self-end{align-self:flex-end!important}.align-self-center{align-self:center!important}.align-self-baseline{align-self:baseline!important}.align-self-stretch{align-self:stretch!important}@media (min-width:576px){.flex-sm-row{flex-direction:row!important}.flex-sm-column{flex-direction:column!important}.flex-sm-row-reverse{flex-direction:row-reverse!important}.flex-sm-column-reverse{flex-direction:column-reverse!important}.flex-sm-wrap{flex-wrap:wrap!important}.flex-sm-nowrap{flex-wrap:nowrap!important}.flex-sm-wrap-reverse{flex-wrap:wrap-reverse!important}.flex-sm-fill{flex:1 1 auto!important}.flex-sm-grow-0{flex-grow:0!important}.flex-sm-grow-1{flex-grow:1!important}.flex-sm-shrink-0{flex-shrink:0!important}.flex-sm-shrink-1{flex-shrink:1!important}.justify-content-sm-start{justify-content:flex-start!important}.justify-content-sm-end{justify-content:flex-end!important}.justify-content-sm-center{justify-content:center!important}.justify-content-sm-between{justify-content:space-between!important}.justify-content-sm-around{justify-content:space-around!important}.align-items-sm-start{align-items:flex-start!important}.align-items-sm-end{align-items:flex-end!important}.align-items-sm-center{align-items:center!important}.align-items-sm-baseline{align-items:baseline!important}.align-items-sm-stretch{align-items:stretch!important}.align-content-sm-start{align-content:flex-start!important}.align-content-sm-end{align-content:flex-end!important}.align-content-sm-center{align-content:center!important}.align-content-sm-between{align-content:space-between!important}.align-content-sm-around{align-content:space-around!important}.align-content-sm-stretch{align-content:stretch!important}.align-self-sm-auto{align-self:auto!important}.align-self-sm-start{align-self:flex-start!important}.align-self-sm-end{align-self:flex-end!important}.align-self-sm-center{align-self:center!important}.align-self-sm-baseline{align-self:baseline!important}.align-self-sm-stretch{align-self:stretch!important}}@media (min-width:768px){.flex-md-row{flex-direction:row!important}.flex-md-column{flex-direction:column!important}.flex-md-row-reverse{flex-direction:row-reverse!important}.flex-md-column-reverse{flex-direction:column-reverse!important}.flex-md-wrap{flex-wrap:wrap!important}.flex-md-nowrap{flex-wrap:nowrap!important}.flex-md-wrap-reverse{flex-wrap:wrap-reverse!important}.flex-md-fill{flex:1 1 auto!important}.flex-md-grow-0{flex-grow:0!important}.flex-md-grow-1{flex-grow:1!important}.flex-md-shrink-0{flex-shrink:0!important}.flex-md-shrink-1{flex-shrink:1!important}.justify-content-md-start{justify-content:flex-start!important}.justify-content-md-end{justify-content:flex-end!important}.justify-content-md-center{justify-content:center!important}.justify-content-md-between{justify-content:space-between!important}.justify-content-md-around{justify-content:space-around!important}.align-items-md-start{align-items:flex-start!important}.align-items-md-end{align-items:flex-end!important}.align-items-md-center{align-items:center!important}.align-items-md-baseline{align-items:baseline!important}.align-items-md-stretch{align-items:stretch!important}.align-content-md-start{align-content:flex-start!important}.align-content-md-end{align-content:flex-end!important}.align-content-md-center{align-content:center!important}.align-content-md-between{align-content:space-between!important}.align-content-md-around{align-content:space-around!important}.align-content-md-stretch{align-content:stretch!important}.align-self-md-auto{align-self:auto!important}.align-self-md-start{align-self:flex-start!important}.align-self-md-end{align-self:flex-end!important}.align-self-md-center{align-self:center!important}.align-self-md-baseline{align-self:baseline!important}.align-self-md-stretch{align-self:stretch!important}}@media (min-width:992px){.flex-lg-row{flex-direction:row!important}.flex-lg-column{flex-direction:column!important}.flex-lg-row-reverse{flex-direction:row-reverse!important}.flex-lg-column-reverse{flex-direction:column-reverse!important}.flex-lg-wrap{flex-wrap:wrap!important}.flex-lg-nowrap{flex-wrap:nowrap!important}.flex-lg-wrap-reverse{flex-wrap:wrap-reverse!important}.flex-lg-fill{flex:1 1 auto!important}.flex-lg-grow-0{flex-grow:0!important}.flex-lg-grow-1{flex-grow:1!important}.flex-lg-shrink-0{flex-shrink:0!important}.flex-lg-shrink-1{flex-shrink:1!important}.justify-content-lg-start{justify-content:flex-start!important}.justify-content-lg-end{justify-content:flex-end!important}.justify-content-lg-center{justify-content:center!important}.justify-content-lg-between{justify-content:space-between!important}.justify-content-lg-around{justify-content:space-around!important}.align-items-lg-start{align-items:flex-start!important}.align-items-lg-end{align-items:flex-end!important}.align-items-lg-center{align-items:center!important}.align-items-lg-baseline{align-items:baseline!important}.align-items-lg-stretch{align-items:stretch!important}.align-content-lg-start{align-content:flex-start!important}.align-content-lg-end{align-content:flex-end!important}.align-content-lg-center{align-content:center!important}.align-content-lg-between{align-content:space-between!important}.align-content-lg-around{align-content:space-around!important}.align-content-lg-stretch{align-content:stretch!important}.align-self-lg-auto{align-self:auto!important}.align-self-lg-start{align-self:flex-start!important}.align-self-lg-end{align-self:flex-end!important}.align-self-lg-center{align-self:center!important}.align-self-lg-baseline{align-self:baseline!important}.align-self-lg-stretch{align-self:stretch!important}}@media (min-width:1200px){.flex-xl-row{flex-direction:row!important}.flex-xl-column{flex-direction:column!important}.flex-xl-row-reverse{flex-direction:row-reverse!important}.flex-xl-column-reverse{flex-direction:column-reverse!important}.flex-xl-wrap{flex-wrap:wrap!important}.flex-xl-nowrap{flex-wrap:nowrap!important}.flex-xl-wrap-reverse{flex-wrap:wrap-reverse!important}.flex-xl-fill{flex:1 1 auto!important}.flex-xl-grow-0{flex-grow:0!important}.flex-xl-grow-1{flex-grow:1!important}.flex-xl-shrink-0{flex-shrink:0!important}.flex-xl-shrink-1{flex-shrink:1!important}.justify-content-xl-start{justify-content:flex-start!important}.justify-content-xl-end{justify-content:flex-end!important}.justify-content-xl-center{justify-content:center!important}.justify-content-xl-between{justify-content:space-between!important}.justify-content-xl-around{justify-content:space-around!important}.align-items-xl-start{align-items:flex-start!important}.align-items-xl-end{align-items:flex-end!important}.align-items-xl-center{align-items:center!important}.align-items-xl-baseline{align-items:baseline!important}.align-items-xl-stretch{align-items:stretch!important}.align-content-xl-start{align-content:flex-start!important}.align-content-xl-end{align-content:flex-end!important}.align-content-xl-center{align-content:center!important}.align-content-xl-between{align-content:space-between!important}.align-content-xl-around{align-content:space-around!important}.align-content-xl-stretch{align-content:stretch!important}.align-self-xl-auto{align-self:auto!important}.align-self-xl-start{align-self:flex-start!important}.align-self-xl-end{align-self:flex-end!important}.align-self-xl-center{align-self:center!important}.align-self-xl-baseline{align-self:baseline!important}.align-self-xl-stretch{align-self:stretch!important}}.float-left{float:left!important}.float-right{float:right!important}.float-none{float:none!important}@media (min-width:576px){.float-sm-left{float:left!important}.float-sm-right{float:right!important}.float-sm-none{float:none!important}}@media (min-width:768px){.float-md-left{float:left!important}.float-md-right{float:right!important}.float-md-none{float:none!important}}@media (min-width:992px){.float-lg-left{float:left!important}.float-lg-right{float:right!important}.float-lg-none{float:none!important}}@media (min-width:1200px){.float-xl-left{float:left!important}.float-xl-right{float:right!important}.float-xl-none{float:none!important}}.user-select-all{user-select:all!important}.user-select-auto{user-select:auto!important}.user-select-none{user-select:none!important}.overflow-auto{overflow:auto!important}.overflow-hidden{overflow:hidden!important}.position-static{position:static!important}.position-relative{position:relative!important}.position-absolute{position:absolute!important}.position-fixed{position:fixed!important}.position-sticky{position:sticky!important}.fixed-top{top:0}.fixed-bottom,.fixed-top{position:fixed;right:0;left:0;z-index:1030}.fixed-bottom{bottom:0}@supports (position:sticky){.sticky-top{position:sticky;top:0;z-index:1020}}.sr-only{position:absolute;width:1px;height:1px;padding:0;margin:-1px;overflow:hidden;clip:rect(0,0,0,0);white-space:nowrap;border:0}.sr-only-focusable:active,.sr-only-focusable:focus{position:static;width:auto;height:auto;overflow:visible;clip:auto;white-space:normal}.shadow-sm{box-shadow:0 .125rem .25rem rgba(0,0,0,.075)!important}.shadow{box-shadow:0 .5rem 1rem rgba(0,0,0,.15)!important}.shadow-lg{box-shadow:0 1rem 3rem rgba(0,0,0,.175)!important}.shadow-none{box-shadow:none!important}.w-25{width:25%!important}.w-50{width:50%!important}.w-75{width:75%!important}.w-100{width:100%!important}.w-auto{width:auto!important}.h-25{height:25%!important}.h-50{height:50%!important}.h-75{height:75%!important}.h-100{height:100%!important}.h-auto{height:auto!important}.mw-100{max-width:100%!important}.mh-100{max-height:100%!important}.min-vw-100{min-width:100vw!important}.min-vh-100{min-height:100vh!important}.vw-100{width:100vw!important}.vh-100{height:100vh!important}.m-0{margin:0!important}.mt-0,.my-0{margin-top:0!important}.mr-0,.mx-0{margin-right:0!important}.mb-0,.my-0{margin-bottom:0!important}.ml-0,.mx-0{margin-left:0!important}.m-1{margin:.25rem!important}.mt-1,.my-1{margin-top:.25rem!important}.mr-1,.mx-1{margin-right:.25rem!important}.mb-1,.my-1{margin-bottom:.25rem!important}.ml-1,.mx-1{margin-left:.25rem!important}.m-2{margin:.5rem!important}.mt-2,.my-2{margin-top:.5rem!important}.mr-2,.mx-2{margin-right:.5rem!important}.mb-2,.my-2{margin-bottom:.5rem!important}.ml-2,.mx-2{margin-left:.5rem!important}.m-3{margin:1rem!important}.mt-3,.my-3{margin-top:1rem!important}.mr-3,.mx-3{margin-right:1rem!important}.mb-3,.my-3{margin-bottom:1rem!important}.ml-3,.mx-3{margin-left:1rem!important}.m-4{margin:1.5rem!important}.mt-4,.my-4{margin-top:1.5rem!important}.mr-4,.mx-4{margin-right:1.5rem!important}.mb-4,.my-4{margin-bottom:1.5rem!important}.ml-4,.mx-4{margin-left:1.5rem!important}.m-5{margin:3rem!important}.mt-5,.my-5{margin-top:3rem!important}.mr-5,.mx-5{margin-right:3rem!important}.mb-5,.my-5{margin-bottom:3rem!important}.ml-5,.mx-5{margin-left:3rem!important}.p-0{padding:0!important}.pt-0,.py-0{padding-top:0!important}.pr-0,.px-0{padding-right:0!important}.pb-0,.py-0{padding-bottom:0!important}.pl-0,.px-0{padding-left:0!important}.p-1{padding:.25rem!important}.pt-1,.py-1{padding-top:.25rem!important}.pr-1,.px-1{padding-right:.25rem!important}.pb-1,.py-1{padding-bottom:.25rem!important}.pl-1,.px-1{padding-left:.25rem!important}.p-2{padding:.5rem!important}.pt-2,.py-2{padding-top:.5rem!important}.pr-2,.px-2{padding-right:.5rem!important}.pb-2,.py-2{padding-bottom:.5rem!important}.pl-2,.px-2{padding-left:.5rem!important}.p-3{padding:1rem!important}.pt-3,.py-3{padding-top:1rem!important}.pr-3,.px-3{padding-right:1rem!important}.pb-3,.py-3{padding-bottom:1rem!important}.pl-3,.px-3{padding-left:1rem!important}.p-4{padding:1.5rem!important}.pt-4,.py-4{padding-top:1.5rem!important}.pr-4,.px-4{padding-right:1.5rem!important}.pb-4,.py-4{padding-bottom:1.5rem!important}.pl-4,.px-4{padding-left:1.5rem!important}.p-5{padding:3rem!important}.pt-5,.py-5{padding-top:3rem!important}.pr-5,.px-5{padding-right:3rem!important}.pb-5,.py-5{padding-bottom:3rem!important}.pl-5,.px-5{padding-left:3rem!important}.m-n1{margin:-.25rem!important}.mt-n1,.my-n1{margin-top:-.25rem!important}.mr-n1,.mx-n1{margin-right:-.25rem!important}.mb-n1,.my-n1{margin-bottom:-.25rem!important}.ml-n1,.mx-n1{margin-left:-.25rem!important}.m-n2{margin:-.5rem!important}.mt-n2,.my-n2{margin-top:-.5rem!important}.mr-n2,.mx-n2{margin-right:-.5rem!important}.mb-n2,.my-n2{margin-bottom:-.5rem!important}.ml-n2,.mx-n2{margin-left:-.5rem!important}.m-n3{margin:-1rem!important}.mt-n3,.my-n3{margin-top:-1rem!important}.mr-n3,.mx-n3{margin-right:-1rem!important}.mb-n3,.my-n3{margin-bottom:-1rem!important}.ml-n3,.mx-n3{margin-left:-1rem!important}.m-n4{margin:-1.5rem!important}.mt-n4,.my-n4{margin-top:-1.5rem!important}.mr-n4,.mx-n4{margin-right:-1.5rem!important}.mb-n4,.my-n4{margin-bottom:-1.5rem!important}.ml-n4,.mx-n4{margin-left:-1.5rem!important}.m-n5{margin:-3rem!important}.mt-n5,.my-n5{margin-top:-3rem!important}.mr-n5,.mx-n5{margin-right:-3rem!important}.mb-n5,.my-n5{margin-bottom:-3rem!important}.ml-n5,.mx-n5{margin-left:-3rem!important}.m-auto{margin:auto!important}.mt-auto,.my-auto{margin-top:auto!important}.mr-auto,.mx-auto{margin-right:auto!important}.mb-auto,.my-auto{margin-bottom:auto!important}.ml-auto,.mx-auto{margin-left:auto!important}@media (min-width:576px){.m-sm-0{margin:0!important}.mt-sm-0,.my-sm-0{margin-top:0!important}.mr-sm-0,.mx-sm-0{margin-right:0!important}.mb-sm-0,.my-sm-0{margin-bottom:0!important}.ml-sm-0,.mx-sm-0{margin-left:0!important}.m-sm-1{margin:.25rem!important}.mt-sm-1,.my-sm-1{margin-top:.25rem!important}.mr-sm-1,.mx-sm-1{margin-right:.25rem!important}.mb-sm-1,.my-sm-1{margin-bottom:.25rem!important}.ml-sm-1,.mx-sm-1{margin-left:.25rem!important}.m-sm-2{margin:.5rem!important}.mt-sm-2,.my-sm-2{margin-top:.5rem!important}.mr-sm-2,.mx-sm-2{margin-right:.5rem!important}.mb-sm-2,.my-sm-2{margin-bottom:.5rem!important}.ml-sm-2,.mx-sm-2{margin-left:.5rem!important}.m-sm-3{margin:1rem!important}.mt-sm-3,.my-sm-3{margin-top:1rem!important}.mr-sm-3,.mx-sm-3{margin-right:1rem!important}.mb-sm-3,.my-sm-3{margin-bottom:1rem!important}.ml-sm-3,.mx-sm-3{margin-left:1rem!important}.m-sm-4{margin:1.5rem!important}.mt-sm-4,.my-sm-4{margin-top:1.5rem!important}.mr-sm-4,.mx-sm-4{margin-right:1.5rem!important}.mb-sm-4,.my-sm-4{margin-bottom:1.5rem!important}.ml-sm-4,.mx-sm-4{margin-left:1.5rem!important}.m-sm-5{margin:3rem!important}.mt-sm-5,.my-sm-5{margin-top:3rem!important}.mr-sm-5,.mx-sm-5{margin-right:3rem!important}.mb-sm-5,.my-sm-5{margin-bottom:3rem!important}.ml-sm-5,.mx-sm-5{margin-left:3rem!important}.p-sm-0{padding:0!important}.pt-sm-0,.py-sm-0{padding-top:0!important}.pr-sm-0,.px-sm-0{padding-right:0!important}.pb-sm-0,.py-sm-0{padding-bottom:0!important}.pl-sm-0,.px-sm-0{padding-left:0!important}.p-sm-1{padding:.25rem!important}.pt-sm-1,.py-sm-1{padding-top:.25rem!important}.pr-sm-1,.px-sm-1{padding-right:.25rem!important}.pb-sm-1,.py-sm-1{padding-bottom:.25rem!important}.pl-sm-1,.px-sm-1{padding-left:.25rem!important}.p-sm-2{padding:.5rem!important}.pt-sm-2,.py-sm-2{padding-top:.5rem!important}.pr-sm-2,.px-sm-2{padding-right:.5rem!important}.pb-sm-2,.py-sm-2{padding-bottom:.5rem!important}.pl-sm-2,.px-sm-2{padding-left:.5rem!important}.p-sm-3{padding:1rem!important}.pt-sm-3,.py-sm-3{padding-top:1rem!important}.pr-sm-3,.px-sm-3{padding-right:1rem!important}.pb-sm-3,.py-sm-3{padding-bottom:1rem!important}.pl-sm-3,.px-sm-3{padding-left:1rem!important}.p-sm-4{padding:1.5rem!important}.pt-sm-4,.py-sm-4{padding-top:1.5rem!important}.pr-sm-4,.px-sm-4{padding-right:1.5rem!important}.pb-sm-4,.py-sm-4{padding-bottom:1.5rem!important}.pl-sm-4,.px-sm-4{padding-left:1.5rem!important}.p-sm-5{padding:3rem!important}.pt-sm-5,.py-sm-5{padding-top:3rem!important}.pr-sm-5,.px-sm-5{padding-right:3rem!important}.pb-sm-5,.py-sm-5{padding-bottom:3rem!important}.pl-sm-5,.px-sm-5{padding-left:3rem!important}.m-sm-n1{margin:-.25rem!important}.mt-sm-n1,.my-sm-n1{margin-top:-.25rem!important}.mr-sm-n1,.mx-sm-n1{margin-right:-.25rem!important}.mb-sm-n1,.my-sm-n1{margin-bottom:-.25rem!important}.ml-sm-n1,.mx-sm-n1{margin-left:-.25rem!important}.m-sm-n2{margin:-.5rem!important}.mt-sm-n2,.my-sm-n2{margin-top:-.5rem!important}.mr-sm-n2,.mx-sm-n2{margin-right:-.5rem!important}.mb-sm-n2,.my-sm-n2{margin-bottom:-.5rem!important}.ml-sm-n2,.mx-sm-n2{margin-left:-.5rem!important}.m-sm-n3{margin:-1rem!important}.mt-sm-n3,.my-sm-n3{margin-top:-1rem!important}.mr-sm-n3,.mx-sm-n3{margin-right:-1rem!important}.mb-sm-n3,.my-sm-n3{margin-bottom:-1rem!important}.ml-sm-n3,.mx-sm-n3{margin-left:-1rem!important}.m-sm-n4{margin:-1.5rem!important}.mt-sm-n4,.my-sm-n4{margin-top:-1.5rem!important}.mr-sm-n4,.mx-sm-n4{margin-right:-1.5rem!important}.mb-sm-n4,.my-sm-n4{margin-bottom:-1.5rem!important}.ml-sm-n4,.mx-sm-n4{margin-left:-1.5rem!important}.m-sm-n5{margin:-3rem!important}.mt-sm-n5,.my-sm-n5{margin-top:-3rem!important}.mr-sm-n5,.mx-sm-n5{margin-right:-3rem!important}.mb-sm-n5,.my-sm-n5{margin-bottom:-3rem!important}.ml-sm-n5,.mx-sm-n5{margin-left:-3rem!important}.m-sm-auto{margin:auto!important}.mt-sm-auto,.my-sm-auto{margin-top:auto!important}.mr-sm-auto,.mx-sm-auto{margin-right:auto!important}.mb-sm-auto,.my-sm-auto{margin-bottom:auto!important}.ml-sm-auto,.mx-sm-auto{margin-left:auto!important}}@media (min-width:768px){.m-md-0{margin:0!important}.mt-md-0,.my-md-0{margin-top:0!important}.mr-md-0,.mx-md-0{margin-right:0!important}.mb-md-0,.my-md-0{margin-bottom:0!important}.ml-md-0,.mx-md-0{margin-left:0!important}.m-md-1{margin:.25rem!important}.mt-md-1,.my-md-1{margin-top:.25rem!important}.mr-md-1,.mx-md-1{margin-right:.25rem!important}.mb-md-1,.my-md-1{margin-bottom:.25rem!important}.ml-md-1,.mx-md-1{margin-left:.25rem!important}.m-md-2{margin:.5rem!important}.mt-md-2,.my-md-2{margin-top:.5rem!important}.mr-md-2,.mx-md-2{margin-right:.5rem!important}.mb-md-2,.my-md-2{margin-bottom:.5rem!important}.ml-md-2,.mx-md-2{margin-left:.5rem!important}.m-md-3{margin:1rem!important}.mt-md-3,.my-md-3{margin-top:1rem!important}.mr-md-3,.mx-md-3{margin-right:1rem!important}.mb-md-3,.my-md-3{margin-bottom:1rem!important}.ml-md-3,.mx-md-3{margin-left:1rem!important}.m-md-4{margin:1.5rem!important}.mt-md-4,.my-md-4{margin-top:1.5rem!important}.mr-md-4,.mx-md-4{margin-right:1.5rem!important}.mb-md-4,.my-md-4{margin-bottom:1.5rem!important}.ml-md-4,.mx-md-4{margin-left:1.5rem!important}.m-md-5{margin:3rem!important}.mt-md-5,.my-md-5{margin-top:3rem!important}.mr-md-5,.mx-md-5{margin-right:3rem!important}.mb-md-5,.my-md-5{margin-bottom:3rem!important}.ml-md-5,.mx-md-5{margin-left:3rem!important}.p-md-0{padding:0!important}.pt-md-0,.py-md-0{padding-top:0!important}.pr-md-0,.px-md-0{padding-right:0!important}.pb-md-0,.py-md-0{padding-bottom:0!important}.pl-md-0,.px-md-0{padding-left:0!important}.p-md-1{padding:.25rem!important}.pt-md-1,.py-md-1{padding-top:.25rem!important}.pr-md-1,.px-md-1{padding-right:.25rem!important}.pb-md-1,.py-md-1{padding-bottom:.25rem!important}.pl-md-1,.px-md-1{padding-left:.25rem!important}.p-md-2{padding:.5rem!important}.pt-md-2,.py-md-2{padding-top:.5rem!important}.pr-md-2,.px-md-2{padding-right:.5rem!important}.pb-md-2,.py-md-2{padding-bottom:.5rem!important}.pl-md-2,.px-md-2{padding-left:.5rem!important}.p-md-3{padding:1rem!important}.pt-md-3,.py-md-3{padding-top:1rem!important}.pr-md-3,.px-md-3{padding-right:1rem!important}.pb-md-3,.py-md-3{padding-bottom:1rem!important}.pl-md-3,.px-md-3{padding-left:1rem!important}.p-md-4{padding:1.5rem!important}.pt-md-4,.py-md-4{padding-top:1.5rem!important}.pr-md-4,.px-md-4{padding-right:1.5rem!important}.pb-md-4,.py-md-4{padding-bottom:1.5rem!important}.pl-md-4,.px-md-4{padding-left:1.5rem!important}.p-md-5{padding:3rem!important}.pt-md-5,.py-md-5{padding-top:3rem!important}.pr-md-5,.px-md-5{padding-right:3rem!important}.pb-md-5,.py-md-5{padding-bottom:3rem!important}.pl-md-5,.px-md-5{padding-left:3rem!important}.m-md-n1{margin:-.25rem!important}.mt-md-n1,.my-md-n1{margin-top:-.25rem!important}.mr-md-n1,.mx-md-n1{margin-right:-.25rem!important}.mb-md-n1,.my-md-n1{margin-bottom:-.25rem!important}.ml-md-n1,.mx-md-n1{margin-left:-.25rem!important}.m-md-n2{margin:-.5rem!important}.mt-md-n2,.my-md-n2{margin-top:-.5rem!important}.mr-md-n2,.mx-md-n2{margin-right:-.5rem!important}.mb-md-n2,.my-md-n2{margin-bottom:-.5rem!important}.ml-md-n2,.mx-md-n2{margin-left:-.5rem!important}.m-md-n3{margin:-1rem!important}.mt-md-n3,.my-md-n3{margin-top:-1rem!important}.mr-md-n3,.mx-md-n3{margin-right:-1rem!important}.mb-md-n3,.my-md-n3{margin-bottom:-1rem!important}.ml-md-n3,.mx-md-n3{margin-left:-1rem!important}.m-md-n4{margin:-1.5rem!important}.mt-md-n4,.my-md-n4{margin-top:-1.5rem!important}.mr-md-n4,.mx-md-n4{margin-right:-1.5rem!important}.mb-md-n4,.my-md-n4{margin-bottom:-1.5rem!important}.ml-md-n4,.mx-md-n4{margin-left:-1.5rem!important}.m-md-n5{margin:-3rem!important}.mt-md-n5,.my-md-n5{margin-top:-3rem!important}.mr-md-n5,.mx-md-n5{margin-right:-3rem!important}.mb-md-n5,.my-md-n5{margin-bottom:-3rem!important}.ml-md-n5,.mx-md-n5{margin-left:-3rem!important}.m-md-auto{margin:auto!important}.mt-md-auto,.my-md-auto{margin-top:auto!important}.mr-md-auto,.mx-md-auto{margin-right:auto!important}.mb-md-auto,.my-md-auto{margin-bottom:auto!important}.ml-md-auto,.mx-md-auto{margin-left:auto!important}}@media (min-width:992px){.m-lg-0{margin:0!important}.mt-lg-0,.my-lg-0{margin-top:0!important}.mr-lg-0,.mx-lg-0{margin-right:0!important}.mb-lg-0,.my-lg-0{margin-bottom:0!important}.ml-lg-0,.mx-lg-0{margin-left:0!important}.m-lg-1{margin:.25rem!important}.mt-lg-1,.my-lg-1{margin-top:.25rem!important}.mr-lg-1,.mx-lg-1{margin-right:.25rem!important}.mb-lg-1,.my-lg-1{margin-bottom:.25rem!important}.ml-lg-1,.mx-lg-1{margin-left:.25rem!important}.m-lg-2{margin:.5rem!important}.mt-lg-2,.my-lg-2{margin-top:.5rem!important}.mr-lg-2,.mx-lg-2{margin-right:.5rem!important}.mb-lg-2,.my-lg-2{margin-bottom:.5rem!important}.ml-lg-2,.mx-lg-2{margin-left:.5rem!important}.m-lg-3{margin:1rem!important}.mt-lg-3,.my-lg-3{margin-top:1rem!important}.mr-lg-3,.mx-lg-3{margin-right:1rem!important}.mb-lg-3,.my-lg-3{margin-bottom:1rem!important}.ml-lg-3,.mx-lg-3{margin-left:1rem!important}.m-lg-4{margin:1.5rem!important}.mt-lg-4,.my-lg-4{margin-top:1.5rem!important}.mr-lg-4,.mx-lg-4{margin-right:1.5rem!important}.mb-lg-4,.my-lg-4{margin-bottom:1.5rem!important}.ml-lg-4,.mx-lg-4{margin-left:1.5rem!important}.m-lg-5{margin:3rem!important}.mt-lg-5,.my-lg-5{margin-top:3rem!important}.mr-lg-5,.mx-lg-5{margin-right:3rem!important}.mb-lg-5,.my-lg-5{margin-bottom:3rem!important}.ml-lg-5,.mx-lg-5{margin-left:3rem!important}.p-lg-0{padding:0!important}.pt-lg-0,.py-lg-0{padding-top:0!important}.pr-lg-0,.px-lg-0{padding-right:0!important}.pb-lg-0,.py-lg-0{padding-bottom:0!important}.pl-lg-0,.px-lg-0{padding-left:0!important}.p-lg-1{padding:.25rem!important}.pt-lg-1,.py-lg-1{padding-top:.25rem!important}.pr-lg-1,.px-lg-1{padding-right:.25rem!important}.pb-lg-1,.py-lg-1{padding-bottom:.25rem!important}.pl-lg-1,.px-lg-1{padding-left:.25rem!important}.p-lg-2{padding:.5rem!important}.pt-lg-2,.py-lg-2{padding-top:.5rem!important}.pr-lg-2,.px-lg-2{padding-right:.5rem!important}.pb-lg-2,.py-lg-2{padding-bottom:.5rem!important}.pl-lg-2,.px-lg-2{padding-left:.5rem!important}.p-lg-3{padding:1rem!important}.pt-lg-3,.py-lg-3{padding-top:1rem!important}.pr-lg-3,.px-lg-3{padding-right:1rem!important}.pb-lg-3,.py-lg-3{padding-bottom:1rem!important}.pl-lg-3,.px-lg-3{padding-left:1rem!important}.p-lg-4{padding:1.5rem!important}.pt-lg-4,.py-lg-4{padding-top:1.5rem!important}.pr-lg-4,.px-lg-4{padding-right:1.5rem!important}.pb-lg-4,.py-lg-4{padding-bottom:1.5rem!important}.pl-lg-4,.px-lg-4{padding-left:1.5rem!important}.p-lg-5{padding:3rem!important}.pt-lg-5,.py-lg-5{padding-top:3rem!important}.pr-lg-5,.px-lg-5{padding-right:3rem!important}.pb-lg-5,.py-lg-5{padding-bottom:3rem!important}.pl-lg-5,.px-lg-5{padding-left:3rem!important}.m-lg-n1{margin:-.25rem!important}.mt-lg-n1,.my-lg-n1{margin-top:-.25rem!important}.mr-lg-n1,.mx-lg-n1{margin-right:-.25rem!important}.mb-lg-n1,.my-lg-n1{margin-bottom:-.25rem!important}.ml-lg-n1,.mx-lg-n1{margin-left:-.25rem!important}.m-lg-n2{margin:-.5rem!important}.mt-lg-n2,.my-lg-n2{margin-top:-.5rem!important}.mr-lg-n2,.mx-lg-n2{margin-right:-.5rem!important}.mb-lg-n2,.my-lg-n2{margin-bottom:-.5rem!important}.ml-lg-n2,.mx-lg-n2{margin-left:-.5rem!important}.m-lg-n3{margin:-1rem!important}.mt-lg-n3,.my-lg-n3{margin-top:-1rem!important}.mr-lg-n3,.mx-lg-n3{margin-right:-1rem!important}.mb-lg-n3,.my-lg-n3{margin-bottom:-1rem!important}.ml-lg-n3,.mx-lg-n3{margin-left:-1rem!important}.m-lg-n4{margin:-1.5rem!important}.mt-lg-n4,.my-lg-n4{margin-top:-1.5rem!important}.mr-lg-n4,.mx-lg-n4{margin-right:-1.5rem!important}.mb-lg-n4,.my-lg-n4{margin-bottom:-1.5rem!important}.ml-lg-n4,.mx-lg-n4{margin-left:-1.5rem!important}.m-lg-n5{margin:-3rem!important}.mt-lg-n5,.my-lg-n5{margin-top:-3rem!important}.mr-lg-n5,.mx-lg-n5{margin-right:-3rem!important}.mb-lg-n5,.my-lg-n5{margin-bottom:-3rem!important}.ml-lg-n5,.mx-lg-n5{margin-left:-3rem!important}.m-lg-auto{margin:auto!important}.mt-lg-auto,.my-lg-auto{margin-top:auto!important}.mr-lg-auto,.mx-lg-auto{margin-right:auto!important}.mb-lg-auto,.my-lg-auto{margin-bottom:auto!important}.ml-lg-auto,.mx-lg-auto{margin-left:auto!important}}@media (min-width:1200px){.m-xl-0{margin:0!important}.mt-xl-0,.my-xl-0{margin-top:0!important}.mr-xl-0,.mx-xl-0{margin-right:0!important}.mb-xl-0,.my-xl-0{margin-bottom:0!important}.ml-xl-0,.mx-xl-0{margin-left:0!important}.m-xl-1{margin:.25rem!important}.mt-xl-1,.my-xl-1{margin-top:.25rem!important}.mr-xl-1,.mx-xl-1{margin-right:.25rem!important}.mb-xl-1,.my-xl-1{margin-bottom:.25rem!important}.ml-xl-1,.mx-xl-1{margin-left:.25rem!important}.m-xl-2{margin:.5rem!important}.mt-xl-2,.my-xl-2{margin-top:.5rem!important}.mr-xl-2,.mx-xl-2{margin-right:.5rem!important}.mb-xl-2,.my-xl-2{margin-bottom:.5rem!important}.ml-xl-2,.mx-xl-2{margin-left:.5rem!important}.m-xl-3{margin:1rem!important}.mt-xl-3,.my-xl-3{margin-top:1rem!important}.mr-xl-3,.mx-xl-3{margin-right:1rem!important}.mb-xl-3,.my-xl-3{margin-bottom:1rem!important}.ml-xl-3,.mx-xl-3{margin-left:1rem!important}.m-xl-4{margin:1.5rem!important}.mt-xl-4,.my-xl-4{margin-top:1.5rem!important}.mr-xl-4,.mx-xl-4{margin-right:1.5rem!important}.mb-xl-4,.my-xl-4{margin-bottom:1.5rem!important}.ml-xl-4,.mx-xl-4{margin-left:1.5rem!important}.m-xl-5{margin:3rem!important}.mt-xl-5,.my-xl-5{margin-top:3rem!important}.mr-xl-5,.mx-xl-5{margin-right:3rem!important}.mb-xl-5,.my-xl-5{margin-bottom:3rem!important}.ml-xl-5,.mx-xl-5{margin-left:3rem!important}.p-xl-0{padding:0!important}.pt-xl-0,.py-xl-0{padding-top:0!important}.pr-xl-0,.px-xl-0{padding-right:0!important}.pb-xl-0,.py-xl-0{padding-bottom:0!important}.pl-xl-0,.px-xl-0{padding-left:0!important}.p-xl-1{padding:.25rem!important}.pt-xl-1,.py-xl-1{padding-top:.25rem!important}.pr-xl-1,.px-xl-1{padding-right:.25rem!important}.pb-xl-1,.py-xl-1{padding-bottom:.25rem!important}.pl-xl-1,.px-xl-1{padding-left:.25rem!important}.p-xl-2{padding:.5rem!important}.pt-xl-2,.py-xl-2{padding-top:.5rem!important}.pr-xl-2,.px-xl-2{padding-right:.5rem!important}.pb-xl-2,.py-xl-2{padding-bottom:.5rem!important}.pl-xl-2,.px-xl-2{padding-left:.5rem!important}.p-xl-3{padding:1rem!important}.pt-xl-3,.py-xl-3{padding-top:1rem!important}.pr-xl-3,.px-xl-3{padding-right:1rem!important}.pb-xl-3,.py-xl-3{padding-bottom:1rem!important}.pl-xl-3,.px-xl-3{padding-left:1rem!important}.p-xl-4{padding:1.5rem!important}.pt-xl-4,.py-xl-4{padding-top:1.5rem!important}.pr-xl-4,.px-xl-4{padding-right:1.5rem!important}.pb-xl-4,.py-xl-4{padding-bottom:1.5rem!important}.pl-xl-4,.px-xl-4{padding-left:1.5rem!important}.p-xl-5{padding:3rem!important}.pt-xl-5,.py-xl-5{padding-top:3rem!important}.pr-xl-5,.px-xl-5{padding-right:3rem!important}.pb-xl-5,.py-xl-5{padding-bottom:3rem!important}.pl-xl-5,.px-xl-5{padding-left:3rem!important}.m-xl-n1{margin:-.25rem!important}.mt-xl-n1,.my-xl-n1{margin-top:-.25rem!important}.mr-xl-n1,.mx-xl-n1{margin-right:-.25rem!important}.mb-xl-n1,.my-xl-n1{margin-bottom:-.25rem!important}.ml-xl-n1,.mx-xl-n1{margin-left:-.25rem!important}.m-xl-n2{margin:-.5rem!important}.mt-xl-n2,.my-xl-n2{margin-top:-.5rem!important}.mr-xl-n2,.mx-xl-n2{margin-right:-.5rem!important}.mb-xl-n2,.my-xl-n2{margin-bottom:-.5rem!important}.ml-xl-n2,.mx-xl-n2{margin-left:-.5rem!important}.m-xl-n3{margin:-1rem!important}.mt-xl-n3,.my-xl-n3{margin-top:-1rem!important}.mr-xl-n3,.mx-xl-n3{margin-right:-1rem!important}.mb-xl-n3,.my-xl-n3{margin-bottom:-1rem!important}.ml-xl-n3,.mx-xl-n3{margin-left:-1rem!important}.m-xl-n4{margin:-1.5rem!important}.mt-xl-n4,.my-xl-n4{margin-top:-1.5rem!important}.mr-xl-n4,.mx-xl-n4{margin-right:-1.5rem!important}.mb-xl-n4,.my-xl-n4{margin-bottom:-1.5rem!important}.ml-xl-n4,.mx-xl-n4{margin-left:-1.5rem!important}.m-xl-n5{margin:-3rem!important}.mt-xl-n5,.my-xl-n5{margin-top:-3rem!important}.mr-xl-n5,.mx-xl-n5{margin-right:-3rem!important}.mb-xl-n5,.my-xl-n5{margin-bottom:-3rem!important}.ml-xl-n5,.mx-xl-n5{margin-left:-3rem!important}.m-xl-auto{margin:auto!important}.mt-xl-auto,.my-xl-auto{margin-top:auto!important}.mr-xl-auto,.mx-xl-auto{margin-right:auto!important}.mb-xl-auto,.my-xl-auto{margin-bottom:auto!important}.ml-xl-auto,.mx-xl-auto{margin-left:auto!important}}.stretched-link:after{position:absolute;top:0;right:0;bottom:0;left:0;z-index:1;pointer-events:auto;content:"";background-color:transparent}.text-monospace{font-family:SFMono-Regular,Menlo,Monaco,Consolas,Liberation Mono,Courier New,monospace!important}.text-justify{text-align:justify!important}.text-wrap{white-space:normal!important}.text-nowrap{white-space:nowrap!important}.text-truncate{overflow:hidden;text-overflow:ellipsis;white-space:nowrap}.text-left{text-align:left!important}.text-right{text-align:right!important}.text-center{text-align:center!important}@media (min-width:576px){.text-sm-left{text-align:left!important}.text-sm-right{text-align:right!important}.text-sm-center{text-align:center!important}}@media (min-width:768px){.text-md-left{text-align:left!important}.text-md-right{text-align:right!important}.text-md-center{text-align:center!important}}@media (min-width:992px){.text-lg-left{text-align:left!important}.text-lg-right{text-align:right!important}.text-lg-center{text-align:center!important}}@media (min-width:1200px){.text-xl-left{text-align:left!important}.text-xl-right{text-align:right!important}.text-xl-center{text-align:center!important}}.text-lowercase{text-transform:lowercase!important}.text-uppercase{text-transform:uppercase!important}.text-capitalize{text-transform:capitalize!important}.font-weight-light{font-weight:300!important}.font-weight-lighter{font-weight:lighter!important}.font-weight-normal{font-weight:400!important}.font-weight-bold{font-weight:700!important}.font-weight-bolder{font-weight:bolder!important}.font-italic{font-style:italic!important}.text-white{color:#fff!important}.text-primary{color:#007bff!important}a.text-primary:focus,a.text-primary:hover{color:#0056b3!important}.text-secondary{color:#6c757d!important}a.text-secondary:focus,a.text-secondary:hover{color:#494f54!important}.text-success{color:#28a745!important}a.text-success:focus,a.text-success:hover{color:#19692c!important}.text-info{color:#17a2b8!important}a.text-info:focus,a.text-info:hover{color:#0f6674!important}.text-warning{color:#ffc107!important}a.text-warning:focus,a.text-warning:hover{color:#ba8b00!important}.text-danger{color:#dc3545!important}a.text-danger:focus,a.text-danger:hover{color:#a71d2a!important}.text-light{color:#f8f9fa!important}a.text-light:focus,a.text-light:hover{color:#cbd3da!important}.text-dark{color:#343a40!important}a.text-dark:focus,a.text-dark:hover{color:#121416!important}.text-body{color:#212529!important}.text-muted{color:#6c757d!important}.text-black-50{color:rgba(0,0,0,.5)!important}.text-white-50{color:hsla(0,0%,100%,.5)!important}.text-hide{font:0/0 a;color:transparent;text-shadow:none;background-color:transparent;border:0}.text-decoration-none{text-decoration:none!important}.text-break{word-wrap:break-word!important}.text-reset{color:inherit!important}.visible{visibility:visible!important}.invisible{visibility:hidden!important}@media print{*,:after,:before{text-shadow:none!important;box-shadow:none!important}a:not(.btn){text-decoration:underline}abbr[title]:after{content:" (" attr(title) ")"}pre{white-space:pre-wrap!important}blockquote,pre{border:1px solid #adb5bd;page-break-inside:avoid}thead{display:table-header-group}img,tr{page-break-inside:avoid}h2,h3,p{orphans:3;widows:3}h2,h3{page-break-after:avoid}@page{size:a3}.container,body{min-width:992px!important}.navbar{display:none}.badge{border:1px solid #000}.table{border-collapse:collapse!important}.table td,.table th{background-color:#fff!important}.table-bordered td,.table-bordered th{border:1px solid #dee2e6!important}.table-dark{color:inherit}.table-dark tbody+tbody,.table-dark td,.table-dark th,.table-dark thead th{border-color:#dee2e6}.table .thead-dark th{color:inherit;border-color:#dee2e6}}html{font-size:15px}body{background-color:#fff;font-family:Lato,sans-serif;font-weight:400;line-height:1.65;color:#333;padding-top:75px}p{margin-bottom:1.15rem;font-size:1em}p.rubric{border-bottom:1px solid #c9c9c9}a{color:#005b81;text-decoration:none}a:hover{color:#e32e00;text-decoration:underline}a.headerlink{color:#c60f0f;font-size:.8em;padding:0 4px;text-decoration:none}a.headerlink:hover{background-color:#c60f0f;color:#fff}.header-style,h1,h2,h3,h4,h5,h6{margin:2.75rem 0 1.05rem;font-family:Open Sans,sans-serif;font-weight:400;line-height:1.15}.header-style:before,h1:before,h2:before,h3:before,h4:before,h5:before,h6:before{display:block;content:"";height:80px;margin:-80px 0 0}h1{margin-top:0;font-size:2.488em}h1,h2{color:#130654}h2{font-size:2.074em}h3{font-size:1.728em}h4{font-size:1.44em}h5{font-size:1.2em}h6{font-size:1em}.text_small,small{font-size:.833em}hr{border:0;border-top:1px solid #e5e5e5}pre{padding:10px;background-color:#fafafa;color:#222;line-height:1.2em;border:1px solid #c9c9c9;margin:1.5em 0;box-shadow:1px 1px 1px #d8d8d8}.navbar{position:fixed}.navbar-brand{position:relative;height:45px;width:auto}.navbar-brand img{max-width:100%;height:100%;width:auto}.navbar-light{background:#fff!important;box-shadow:0 .125rem .25rem 0 rgba(0,0,0,.11)}.navbar-nav li a{padding:0 15px}.navbar-nav>.active>.nav-link{font-weight:600;color:#130654!important}.navbar-header a{padding:0 15px}.admonition{margin:1.5625em auto;padding:0 .6rem .8rem!important;overflow:hidden;page-break-inside:avoid;border-left:.2rem solid #007bff;border-radius:.1rem;box-shadow:0 .2rem .5rem rgba(0,0,0,.05),0 0 .05rem rgba(0,0,0,.1);transition:color .25s,background-color .25s,border-color .25s}.admonition :last-child{margin-bottom:0}.admonition p.admonition-title~*{padding:0 1.4rem}.admonition>ol,.admonition>ul{margin-left:1em}.admonition .admonition-title{position:relative;margin:0 -.6rem!important;padding:.4rem .6rem .4rem 2rem;font-weight:700;background-color:rgba(68,138,255,.1)}.admonition .admonition-title:before{position:absolute;left:.6rem;width:1rem;height:1rem;color:#007bff;font-family:Font Awesome\ 5 Free;font-weight:900;content:""}.admonition .admonition-title+*{margin-top:.4em}.admonition.attention{border-color:#fd7e14}.admonition.attention .admonition-title{background-color:#ffedcc}.admonition.attention .admonition-title:before{color:#fd7e14;content:""}.admonition.caution{border-color:#fd7e14}.admonition.caution .admonition-title{background-color:#ffedcc}.admonition.caution .admonition-title:before{color:#fd7e14;content:""}.admonition.warning{border-color:#dc3545}.admonition.warning .admonition-title{background-color:#fdf3f2}.admonition.warning .admonition-title:before{color:#dc3545;content:""}.admonition.danger{border-color:#dc3545}.admonition.danger .admonition-title{background-color:#fdf3f2}.admonition.danger .admonition-title:before{color:#dc3545;content:""}.admonition.error{border-color:#dc3545}.admonition.error .admonition-title{background-color:#fdf3f2}.admonition.error .admonition-title:before{color:#dc3545;content:""}.admonition.hint{border-color:#ffc107}.admonition.hint .admonition-title{background-color:#fff6dd}.admonition.hint .admonition-title:before{color:#ffc107;content:""}.admonition.tip{border-color:#ffc107}.admonition.tip .admonition-title{background-color:#fff6dd}.admonition.tip .admonition-title:before{color:#ffc107;content:""}.admonition.important{border-color:#007bff}.admonition.important .admonition-title{background-color:#e7f2fa}.admonition.important .admonition-title:before{color:#007bff;content:""}.admonition.note{border-color:#007bff}.admonition.note .admonition-title{background-color:#e7f2fa}.admonition.note .admonition-title:before{color:#007bff;content:""}div.deprecated{margin-bottom:10px;margin-top:10px;padding:7px;color:#b94a48;background-color:#f3e5e5;border:1px solid #eed3d7;border-radius:.5rem}div.deprecated p{display:inline}.topic{background-color:#eee}.seealso dd{margin-top:0;margin-bottom:0}.viewcode-back{font-family:Lato,sans-serif}.viewcode-block:target{background-color:#f4debf;border-top:1px solid #ac9;border-bottom:1px solid #ac9}table.field-list{border-collapse:separate;border-spacing:10px;margin-left:1px}table.field-list th.field-name{padding:1px 8px 1px 5px;white-space:nowrap;background-color:#eee}table.field-list td.field-body p{font-style:italic}table.field-list td.field-body p>strong{font-style:normal}table.field-list td.field-body blockquote{border-left:none;margin:0 0 .3em;padding-left:30px}.table.autosummary td:first-child{white-space:nowrap}.footer{width:100%;border-top:1px solid #ccc;padding-top:10px}.bd-search{position:relative;padding:1rem 15px;margin-right:-15px;margin-left:-15px}.bd-search .icon{position:absolute;color:#a4a6a7;left:25px;top:25px}.bd-search input{border-radius:0;border:0;border-bottom:1px solid #e5e5e5;padding-left:35px}.bd-toc{-ms-flex-order:2;order:2;height:calc(100vh - 2rem);overflow-y:auto}@supports (position:-webkit-sticky) or (position:sticky){.bd-toc{position:-webkit-sticky;position:sticky;top:5rem;height:calc(100vh - 5rem);overflow-y:auto}}.bd-toc .onthispage{color:#a4a6a7}.section-nav{padding-left:0;border-left:1px solid #eee;border-bottom:none}.section-nav ul{padding-left:1rem}.toc-entry,.toc-entry a{display:block}.toc-entry a{padding:.125rem 1.5rem;color:#77757a}@media (min-width:1200px){.toc-entry a{padding-right:0}}.toc-entry a:hover{color:rgba(0,0,0,.85);text-decoration:none}.bd-sidebar{padding-top:1em}@media (min-width:768px){.bd-sidebar{border-right:1px solid rgba(0,0,0,.1)}@supports (position:-webkit-sticky) or (position:sticky){.bd-sidebar{position:-webkit-sticky;position:sticky;top:76px;z-index:1000;height:calc(100vh - 4rem)}}}.bd-links{padding-top:1rem;padding-bottom:1rem;margin-right:-15px;margin-left:-15px}@media (min-width:768px){@supports (position:-webkit-sticky) or (position:sticky){.bd-links{max-height:calc(100vh - 9rem);overflow-y:auto}}}@media (min-width:768px){.bd-links{display:block!important}}.bd-sidenav{display:none}.bd-content{padding-top:20px}.bd-content .section{max-width:100%}.bd-content .section table{display:block;overflow:auto}.bd-toc-link{display:block;padding:.25rem 1.5rem;font-weight:600;color:rgba(0,0,0,.65)}.bd-toc-link:hover{color:rgba(0,0,0,.85);text-decoration:none}.bd-toc-item.active{margin-bottom:1rem}.bd-toc-item.active:not(:first-child){margin-top:1rem}.bd-toc-item.active>.bd-toc-link{color:rgba(0,0,0,.85)}.bd-toc-item.active>.bd-toc-link:hover{background-color:transparent}.bd-toc-item.active>.bd-sidenav{display:block}.bd-sidebar .nav>li>a{display:block;padding:.25rem 1.5rem;font-size:.9em;color:rgba(0,0,0,.65)}.bd-sidebar .nav>li>a:hover{color:#130654;text-decoration:none;background-color:transparent}.bd-sidebar .nav>.active:hover>a,.bd-sidebar .nav>.active>a{font-weight:600;color:#130654}.bd-sidebar .nav>li>ul{list-style:none;padding:.25rem 1.5rem}.bd-sidebar .nav>li>ul>li>a{display:block;padding:.25rem 1.5rem;font-size:.9em;color:rgba(0,0,0,.65)}.bd-sidebar .nav>li>ul>.active:hover>a,.bd-sidebar .nav>li>ul>.active>a{font-weight:600;color:#130654}.toc-h2{font-size:.85rem}.toc-h3{font-size:.75rem}.toc-h4{font-size:.65rem}.toc-entry>.nav-link.active{font-weight:600;color:#130654;background-color:transparent;border-left:2px solid #563d7c}.nav-link:hover{border-style:none}#navbar-main-elements li.nav-item i{font-size:.7rem;padding-left:2px;vertical-align:middle}.bd-toc .nav .nav{display:none}.bd-toc .nav .nav.visible,.bd-toc .nav>.active>ul{display:block}.prev-next-bottom{margin:20px 0}.prev-next-bottom a.left-prev,.prev-next-bottom a.right-next{padding:10px;border:1px solid rgba(0,0,0,.2);max-width:45%;overflow-x:hidden;color:rgba(0,0,0,.65)}.prev-next-bottom a.left-prev{float:left}.prev-next-bottom a.left-prev:before{content:"<< "}.prev-next-bottom a.right-next{float:right}.prev-next-bottom a.right-next:after{content:" >>"}.alert{padding-bottom:0}.alert-info a{color:#e83e8c}i.fab{vertical-align:middle;font-style:normal;font-size:1.5rem;line-height:1.25}i.fa-github-square:before{color:#333}i.fa-twitter-square:before{color:#55acee}.tocsection{border-left:1px solid #eee;padding:.3rem 1.5rem}.tocsection i{padding-right:.5rem}.editthispage{padding-top:2rem}.editthispage a{color:#130754}.xr-wrap[hidden]{display:block!important} \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/_static/doctools.js b/doc/LectureNotes/_build/html/_static/doctools.js new file mode 100644 index 000000000..7d88f807d --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/doctools.js @@ -0,0 +1,316 @@ +/* + * doctools.js + * ~~~~~~~~~~~ + * + * Sphinx JavaScript utilities for all documentation. + * + * :copyright: Copyright 2007-2020 by the Sphinx team, see AUTHORS. + * :license: BSD, see LICENSE for details. + * + */ + +/** + * select a different prefix for underscore + */ +$u = _.noConflict(); + +/** + * make the code below compatible with browsers without + * an installed firebug like debugger +if (!window.console || !console.firebug) { + var names = ["log", "debug", "info", "warn", "error", "assert", "dir", + "dirxml", "group", "groupEnd", "time", "timeEnd", "count", "trace", + "profile", "profileEnd"]; + window.console = {}; + for (var i = 0; i < names.length; ++i) + window.console[names[i]] = function() {}; +} + */ + +/** + * small helper function to urldecode strings + */ +jQuery.urldecode = function(x) { + return decodeURIComponent(x).replace(/\+/g, ' '); +}; + +/** + * small helper function to urlencode strings + */ +jQuery.urlencode = encodeURIComponent; + +/** + * This function returns the parsed url parameters of the + * current request. Multiple values per key are supported, + * it will always return arrays of strings for the value parts. + */ +jQuery.getQueryParameters = function(s) { + if (typeof s === 'undefined') + s = document.location.search; + var parts = s.substr(s.indexOf('?') + 1).split('&'); + var result = {}; + for (var i = 0; i < parts.length; i++) { + var tmp = parts[i].split('=', 2); + var key = jQuery.urldecode(tmp[0]); + var value = jQuery.urldecode(tmp[1]); + if (key in result) + result[key].push(value); + else + result[key] = [value]; + } + return result; +}; + +/** + * highlight a given string on a jquery object by wrapping it in + * span elements with the given class name. + */ +jQuery.fn.highlightText = function(text, className) { + function highlight(node, addItems) { + if (node.nodeType === 3) { + var val = node.nodeValue; + var pos = val.toLowerCase().indexOf(text); + if (pos >= 0 && + !jQuery(node.parentNode).hasClass(className) && + !jQuery(node.parentNode).hasClass("nohighlight")) { + var span; + var isInSVG = jQuery(node).closest("body, svg, foreignObject").is("svg"); + if (isInSVG) { + span = document.createElementNS("http://www.w3.org/2000/svg", "tspan"); + } else { + span = document.createElement("span"); + span.className = className; + } + span.appendChild(document.createTextNode(val.substr(pos, text.length))); + node.parentNode.insertBefore(span, node.parentNode.insertBefore( + document.createTextNode(val.substr(pos + text.length)), + node.nextSibling)); + node.nodeValue = val.substr(0, pos); + if (isInSVG) { + var rect = document.createElementNS("http://www.w3.org/2000/svg", "rect"); + var bbox = node.parentElement.getBBox(); + rect.x.baseVal.value = bbox.x; + rect.y.baseVal.value = bbox.y; + rect.width.baseVal.value = bbox.width; + rect.height.baseVal.value = bbox.height; + rect.setAttribute('class', className); + addItems.push({ + "parent": node.parentNode, + "target": rect}); + } + } + } + else if (!jQuery(node).is("button, select, textarea")) { + jQuery.each(node.childNodes, function() { + highlight(this, addItems); + }); + } + } + var addItems = []; + var result = this.each(function() { + highlight(this, addItems); + }); + for (var i = 0; i < addItems.length; ++i) { + jQuery(addItems[i].parent).before(addItems[i].target); + } + return result; +}; + +/* + * backward compatibility for jQuery.browser + * This will be supported until firefox bug is fixed. + */ +if (!jQuery.browser) { + jQuery.uaMatch = function(ua) { + ua = ua.toLowerCase(); + + var match = /(chrome)[ \/]([\w.]+)/.exec(ua) || + /(webkit)[ \/]([\w.]+)/.exec(ua) || + /(opera)(?:.*version|)[ \/]([\w.]+)/.exec(ua) || + /(msie) ([\w.]+)/.exec(ua) || + ua.indexOf("compatible") < 0 && /(mozilla)(?:.*? rv:([\w.]+)|)/.exec(ua) || + []; + + return { + browser: match[ 1 ] || "", + version: match[ 2 ] || "0" + }; + }; + jQuery.browser = {}; + jQuery.browser[jQuery.uaMatch(navigator.userAgent).browser] = true; +} + +/** + * Small JavaScript module for the documentation. + */ +var Documentation = { + + init : function() { + this.fixFirefoxAnchorBug(); + this.highlightSearchWords(); + this.initIndexTable(); + if (DOCUMENTATION_OPTIONS.NAVIGATION_WITH_KEYS) { + this.initOnKeyListeners(); + } + }, + + /** + * i18n support + */ + TRANSLATIONS : {}, + PLURAL_EXPR : function(n) { return n === 1 ? 0 : 1; }, + LOCALE : 'unknown', + + // gettext and ngettext don't access this so that the functions + // can safely bound to a different name (_ = Documentation.gettext) + gettext : function(string) { + var translated = Documentation.TRANSLATIONS[string]; + if (typeof translated === 'undefined') + return string; + return (typeof translated === 'string') ? translated : translated[0]; + }, + + ngettext : function(singular, plural, n) { + var translated = Documentation.TRANSLATIONS[singular]; + if (typeof translated === 'undefined') + return (n == 1) ? singular : plural; + return translated[Documentation.PLURALEXPR(n)]; + }, + + addTranslations : function(catalog) { + for (var key in catalog.messages) + this.TRANSLATIONS[key] = catalog.messages[key]; + this.PLURAL_EXPR = new Function('n', 'return +(' + catalog.plural_expr + ')'); + this.LOCALE = catalog.locale; + }, + + /** + * add context elements like header anchor links + */ + addContextElements : function() { + $('div[id] > :header:first').each(function() { + $('\u00B6'). + attr('href', '#' + this.id). + attr('title', _('Permalink to this headline')). + appendTo(this); + }); + $('dt[id]').each(function() { + $('\u00B6'). + attr('href', '#' + this.id). + attr('title', _('Permalink to this definition')). + appendTo(this); + }); + }, + + /** + * workaround a firefox stupidity + * see: https://bugzilla.mozilla.org/show_bug.cgi?id=645075 + */ + fixFirefoxAnchorBug : function() { + if (document.location.hash && $.browser.mozilla) + window.setTimeout(function() { + document.location.href += ''; + }, 10); + }, + + /** + * highlight the search words provided in the url in the text + */ + highlightSearchWords : function() { + var params = $.getQueryParameters(); + var terms = (params.highlight) ? params.highlight[0].split(/\s+/) : []; + if (terms.length) { + var body = $('div.body'); + if (!body.length) { + body = $('body'); + } + window.setTimeout(function() { + $.each(terms, function() { + body.highlightText(this.toLowerCase(), 'highlighted'); + }); + }, 10); + $('') + .appendTo($('#searchbox')); + } + }, + + /** + * init the domain index toggle buttons + */ + initIndexTable : function() { + var togglers = $('img.toggler').click(function() { + var src = $(this).attr('src'); + var idnum = $(this).attr('id').substr(7); + $('tr.cg-' + idnum).toggle(); + if (src.substr(-9) === 'minus.png') + $(this).attr('src', src.substr(0, src.length-9) + 'plus.png'); + else + $(this).attr('src', src.substr(0, src.length-8) + 'minus.png'); + }).css('display', ''); + if (DOCUMENTATION_OPTIONS.COLLAPSE_INDEX) { + togglers.click(); + } + }, + + /** + * helper function to hide the search marks again + */ + hideSearchWords : function() { + $('#searchbox .highlight-link').fadeOut(300); + $('span.highlighted').removeClass('highlighted'); + }, + + /** + * make the url absolute + */ + makeURL : function(relativeURL) { + return DOCUMENTATION_OPTIONS.URL_ROOT + '/' + relativeURL; + }, + + /** + * get the current relative url + */ + getCurrentURL : function() { + var path = document.location.pathname; + var parts = path.split(/\//); + $.each(DOCUMENTATION_OPTIONS.URL_ROOT.split(/\//), function() { + if (this === '..') + parts.pop(); + }); + var url = parts.join('/'); + return path.substring(url.lastIndexOf('/') + 1, path.length - 1); + }, + + initOnKeyListeners: function() { + $(document).keydown(function(event) { + var activeElementType = document.activeElement.tagName; + // don't navigate when in search box, textarea, dropdown or button + if (activeElementType !== 'TEXTAREA' && activeElementType !== 'INPUT' && activeElementType !== 'SELECT' + && activeElementType !== 'BUTTON' && !event.altKey && !event.ctrlKey && !event.metaKey + && !event.shiftKey) { + switch (event.keyCode) { + case 37: // left + var prevHref = $('link[rel="prev"]').prop('href'); + if (prevHref) { + window.location.href = prevHref; + return false; + } + case 39: // right + var nextHref = $('link[rel="next"]').prop('href'); + if (nextHref) { + window.location.href = nextHref; + return false; + } + } + } + }); + } +}; + +// quick alias for translations +_ = Documentation.gettext; + +$(document).ready(function() { + Documentation.init(); +}); diff --git a/doc/LectureNotes/_build/html/_static/documentation_options.js b/doc/LectureNotes/_build/html/_static/documentation_options.js new file mode 100644 index 000000000..93b7c24d6 --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/documentation_options.js @@ -0,0 +1,12 @@ +var DOCUMENTATION_OPTIONS = { + URL_ROOT: document.getElementById("documentation_options").getAttribute('data-url_root'), + VERSION: '', + LANGUAGE: 'None', + COLLAPSE_INDEX: false, + BUILDER: 'html', + FILE_SUFFIX: '.html', + LINK_SUFFIX: '.html', + HAS_SOURCE: true, + SOURCELINK_SUFFIX: '', + NAVIGATION_WITH_KEYS: true +}; \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/_static/file.png b/doc/LectureNotes/_build/html/_static/file.png new file mode 100644 index 0000000000000000000000000000000000000000..a858a410e4faa62ce324d814e4b816fff83a6fb3 GIT binary patch literal 286 zcmV+(0pb3MP)s`hMrGg#P~ix$^RISR_I47Y|r1 z_CyJOe}D1){SET-^Amu_i71Lt6eYfZjRyw@I6OQAIXXHDfiX^GbOlHe=Ae4>0m)d(f|Me07*qoM6N<$f}vM^LjV8( literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/html/_static/images/logo_binder.svg b/doc/LectureNotes/_build/html/_static/images/logo_binder.svg new file mode 100644 index 000000000..45fecf751 --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/images/logo_binder.svg @@ -0,0 +1,19 @@ + + + + +logo + + + + + + + + diff --git a/doc/LectureNotes/_build/html/_static/images/logo_colab.png b/doc/LectureNotes/_build/html/_static/images/logo_colab.png new file mode 100644 index 0000000000000000000000000000000000000000..b7560ec216b2d1b6f77855525fe966c741833428 GIT binary patch literal 7601 zcmeI1^;ZuSFsz@@e&Hu|o~yU_Jn_7Cy4b4(M?f2S`owL6D#ysoM3Rsb4MX|l6hl52QIsX*kmQMmFZ6Xu|Wk1r15+E^+Er?@^MFpIE zq!=C|$Nn*F4aR@N|DPxS6E^f|7Z=H%T>vS)_|-RkkprWw zSGb9TlwheKfo{U5J)kX1$cHtEFe}Pa2Au|?^hCk%8gdI}l*ypIUsLXLMy9W|q-ZAw zJpZkmGRa|!=7CyrA#Bs2?5UdZ1^pDaji}+DimdE$JB@FrJvAIxy*3v#1-8OwO;OS$ zsv*P<%V4%?*Keca@o9}LMOs~ph)z!AU;${{23k&Gq7A@nDP{*I1HiTZ=Q*54?Bok) zp6L_4HhiE->YU6{m*{7O7j#SkBb9JPo!k8TD0H6{ zdSE-mmA!Js{}(?qh${0wB7Rx{*F=43D>?j3kU8MX&`sQJ+wHUD6eEr7j%*2x%5|a8 z*;AP<*tCQwj`Af5vvGHXF=9{cdzV2BMI@}VHgmol)^f>Ectcls5p3dW?40~ADd>ki za*q>v=nQQmGI5&BS!GU|iX9>qB9r=_Qm9t_Qwi+zWI zc%%oQ`P}{ZXk^}?+H!u2my^C#TD%=V|3pb$MXhJ07bx-^=oxj?ZSk!---?f2cs8_& z8?O{lvxMDZi7gsdvoZ2bmyLYs1!O1RMC)1Wv`9p-I(1pfww9siX;Lu>^>_Y=g+OHo zPm(N|h?h5Z>yze~wKtPBRv(mZx*A4R%bganw#OV=SE*=J^b#~(YfIcj(k=(i37PY7 zUiawSj8SKczPk-^=SwOOb%X+bRcFm+=N1r{{CA<=kbVq8cFGcLSGqM5FUxChbc&`o9$mUo4kZLh+%KP6m zDMd3SH~N5fH8J+8;bpxhi-9i}^PV(^u?zb49_c!Ow_!1w%w(RLEeXJoMU>Nnlc8sd z<;K$L<-WwC`NJ0PWzB59Pzbg|FZS-=xlaWDjM-PXIJ;r4qyFnFc_<-VDg5P=Zk0Pd z%f7GFg?FzC??rmjG^Ib<{cfE+dud-%)Ep=a8Q(Z-Fng}&CvD+JPdO)mL-$u4eH#LJ z7heze_GA*{rYAL;ejb#P;oTD_*Rgrw;)1(e;+zGN{)D)k?o$t&BGWEM!Hn}LQm1jd zf@B0+pEzI&qREI@Qr=#K;u~Fs)Saf>_1X|EQGz0D_a|>)d?IOck($^4a`v4Hc6sKV zgm7-VK|sz+(A$-L0BnhZ#qKk${svcv4#QmCcMCb>t9=e+^b49rrK@5C@-Qs{PN6H8Tb^nIy#)VA`)o~+c~m2m9bN}EcwI`-IP+fB&d^;19iX9{XvM6VYHE(fX{BIU zjMLmkl7p}TslG;@C!HvX=7hVy6cGIM{h7hxrM^q{j`Y4Ux1nI*k9MB?ToSK!Qpvy< zT~`Qofe|OBk8vza_r02Y;~+V6WKn(J{_?BR9@-`D&Q;nTEx7+j36Qk0(l3TahUki} z;O-FUuOnNVcc-Q3c?;A)ZpgKC-Sa8`{c}MNm$j))KPPdL#xR*0kxQz|V-;WZxI+?u zFB#~P=os0);b?+6$-z@yE%k*^!0x)K_!|4!L%ADpXqe`pG|8A+rht_!jZid=wb1j& zjPG_SeS*{ef!h*}~k!*;Aar3`tCeHO@>c{c>ak(x3f^w3+_zT>j)aP_hVoV4~^0L<5^eu_y z-@tf0YyH-(#5uTh`s3DIhpc^`UysO{L8JS|z=qnHFb)UqfMnC!Hu$=eiC+a;9t*X6R?Q8POFRq?_ak1&yP&YF6`@B=qySm8MJ)n*E zdS-&E$a$DMp!}+S%^(Q))m7O$Qece1ZtB+=H{**c0@XT53VGNeFhvnDVocubi6~ru z2X&(|kp)joFLfuG?i;d=&CZBQhez8i+lhV+c;_pEL6+Teo z1qclCF-EO~XWkH3u|unGI79@`+YLi}rF>PbBrn{PBKWF&S%K6N0u^DRx7qImnJ`+c z>Nu)TJyhpyJX_!XHh^82M+YgW&cxs(vQKEpL%}iK(hH=<@)j#E3_?a*JP@0=R z;O*(_2@>IjYLClnL+$PJ-5!vt6>UJ7$KHM3LlFFMxb19oFZ_fi@{fp};$@_n8driG z`=77&{Z^0#T>t%$hCqQi8M}0E4XipxikcsB$>o9M)rBJWQDY7UrgKAy|BP4kr`Nay z??T|Ajh_U=3lem-tL$_tEhB=Rqfi?bUj`u>$a-x5WxqHn6t4)Q-NQ^Bt-k!mcE0ES z4)*3-(5@V)=EloLT~ReorH252&Q&MWWc$oiSS{!xpO?VPpJFD-QN6c=<7HxnH1nH% zeiOM22U=%trq`HCXYNL#H!P!M1{?)QcIGYWO$;mCMHnpgd?*ZE&bmylPxndZ$B}ct zIfSCaCu!a^rBwLoo4gQJnU<%~!6cPP-qxJLZM#F&_gwU%?O$k?DIF6l%q_lvcs3})|Z?z(K3q9(BASQtZlw@+<5mv zrHuRbc}A4I9hLtxbS!@ju49VVt1XxpO?1&$LA;?ZANYo=SC^nMg{9BY`=cZcTaR{A@r{UB@;%H zPb6QWRuvU)J>>*0FB;9Uq|hH4C$u8T=T?sz{5%Ex)I%5W6wQmtel=rJ)Tbw#E7{Z;t3U zY9a$t=WkneF<9867^HBvLp>hs;A@H}9KEwn2t!?ITQ1vZ?fCFF(RfFYplQUymF`y4 z74MX)v7%4i_52G~fn=&qCfo}f%Gj8bd7dI^BDI?AlVN_!qWMJT#NBLs^p)e{tG?D4 z)|x9tIcLpO$-JtVj=#$1Y&GRE*-xUKd_{uxiZkqAudNRF!dph|+p41KtIf(8)c1p~ zv)f(_RGUK*j_{s!DNDET-@ekFNlnTXW_=+4t5>Qbq`aWl%F6e}e)<=0U{Lp}8twQ? z8cJ&^2hntuxcqQ~k;<29cTQz)@X@zbQN?f1q??MK&`gi2me&l@XLSxN|!? z;kRJcy-ahz{?{Aj;b0E9*MKf|Q@H!%2FhB8=t$dhTtR4^%hSctIRz;tXJPme_gd zLiJlhH^x9|I?_vaIKkgiAyrk&%Mv26OqK|av#t%u9aU2`wvZ61wo4$DW%z~d9P`5& zx2Zk{zL$Z1@bGicZ})KZzJKhZaZ+P!-p1uH9dgwUQ5u(q{HyTaprSe95WuIadBYv0 zPUJ~G+G2~n0DfE{7!{N*#1+?ql4nK8`Fr?o@j~3c(>T^^trK4t~7#7WQoVk)7KnFY{iPIQ?Qh8 z+Wy6Ol|m6pA8r4lQdt@$=Z{k}^_evzh~Vt_J$aBM!djok7rTfxt8f+KVv7GM1Awc>b%$6NDX zcl~`@-PYtGJSGIO(C^sr&BxXHz*cUJnB~X1`0$kX)@xH+qFRp1^Vpt^u3V$(w;_vf zHIi3Mb+A5@Nx^>r8g^tF%=j0o$Rhli22c4xiy2SEGE=Dk)m)mzF}VhHtiP43?%dTPKbDg+Gmq$pq6DlCZzY5@`})4DTSfgVh3B z6B#;izoI9B%{^V1qYVp<-KgZ=_(;UqyU^wT{IFPQ?YY4%;yq4cbgN`_dqp${t%ytU z!T>q+J?*26u4Ak4Jx#9uHgScR2!%5YX9%5Bu@HL^VaJ7%jj#ceYuaRZk7vMWX)jq| z-rX)3v33MqZ$qaWp!X$i1yJ*rOfjP-u6noa{n9pxzJw0P2+@UNLHS(-e>##A#9xc` zAr=;dh7~9d71L_&bj`DI@l$2 zSX@4j7tZbUYdo?rgctpAg3>Z@gv1{~grCRQUGVyTbzIJ-YZt2xF(cT)W0~l-76Lw* z<6YF%D4R$X>ZEj#!c)zMi018e@?^1%&N`zutD(OQ;X8am+pNW(YhRwy*%wrsnwb#T z>n{K;55wQE!cVF)X+X12fX<x`lE~DquFsMPRoBuzhuVdR8Gv zevya06i9>q3oJZyDGUHOP=iTbBg`AO7~BI0N8$lqEvK_=V)(Du!8=i|%_2^xqnCgh zYEho!c`8!%;N8>VD_@8NZxuyDHBlxl_=CBT5z4cft(NLsv9Wo81)VnjTne@sFAuLA zv^?3h>Rc?eDzkn@SvwCF^spU#ZJuQz6o4V90>Al2JL^>6N4y0wyg#4m?khQ$4$xa5 zlJZV5E$o~arUalDb_b7lXJs*(UA*P>jQ%3i`I8pyKN?*kY>iRE7J9GGiz^nA>aIV> zaJ}>Ecj_*#d8xFcjhy+6oRGfCr^qR6C2fGkhPUT-of7St?XBEaY>?_o$Y;IiV*<6d zlA;M(1^;P>tJxjiTQAB{T$TKPJ?7HfGON=ms6=%yai0?j-qHB-nhvKj_0=^YawDhO z&$wC;93X#RhmcNJTfn66z&E;UAFGeV6TsD61;r(%GZvUrDg2W3Y2hPsTqkinoI4PV zXDedcq+P^|`+Zqpt5*;9cKbAf6!xI4X{#P5OMaE4?*}B?BIY^Gyv0%UUq}lKO~C#Z zCRamrC=OeXKTKm|4p>}U!kLbE%NxPGuZ1-DR(wWFK@>24ca*qhEt5B*r|(Kty!Pj0 zZauh;NqoiV&&q9pT#S7@dl4JUVA|RmaH8kslFhypJ_)20*ebs^yXIQA(6mi|Wph<8 z=`?$6$QX%TaWE9DLjOgi>rciE+f(9`A4gn4&jZA)v29ug%2=CtvV-U|71pd@edT~> zTA~BLBxs`RYEh%@DuEBdVt=S~6x5VXGkg4=c(|;e@Uk2Mxd}~#h^+`jF}r@=C0+HS zJcg`@*AUj2Ymhzqb=;b}w_oSQ>VH<@k=B`!P>>u5;cpo7O#PB&IQ>AS{06fz5fsXyOt1R0^~JUdht$M7yYTxq$&$T&teFpg;y{BUxXR(00s6bHa2EU zQz~u3(zn7I;Ei{D%kc60jYvUAK^2vZcMr$(Mvo58z}?>{fBdZv&KdKaM(W*WeijQ+ z;}+j>_K=@gAG4KLl-oHs1uHl{4Iq_bV|(|n23Ml=$x+vE+w;rZ1-;Cgwa-{hvjGND zf$}y#wu81ZOPZ@Wj}WbIj4k%PEPTy)sLP0Kk0C=n2lpOrPl~et;FC1`zjD=4!5coL zUgdZMo&inr`+cr#<^beEmG){%LjzXvEJ;=`hMnEYG|VU#W^gR^?uh;u@MsY$78=09EY#xn`@9X5)nb~&t)6wi zB(Y#$oL!o_oI|#`LeD5m>ezV6;nKHq@ZYvUufb~M33Qw%6`GhEa}S@P!}T;dH@bLx zG_yiKDTq6zQz}25>oeWOXpL<9!kJrP)LQASx)Dh$MiaKmk}q7TZJjtiA`M6zv_)Sn zoW-S@(c2ebP+DQqvD-S;#gt=zlveyhax!aybe(eZtlKEO1+bZSMlogo_jupyterhubHub diff --git a/doc/LectureNotes/_build/html/_static/jquery-3.5.1.js b/doc/LectureNotes/_build/html/_static/jquery-3.5.1.js new file mode 100644 index 000000000..50937333b --- /dev/null +++ b/doc/LectureNotes/_build/html/_static/jquery-3.5.1.js @@ -0,0 +1,10872 @@ +/*! + * jQuery JavaScript Library v3.5.1 + * https://jquery.com/ + * + * Includes Sizzle.js + * https://sizzlejs.com/ + * + * Copyright JS Foundation and other contributors + * Released under the MIT license + * https://jquery.org/license + * + * Date: 2020-05-04T22:49Z + */ +( function( global, factory ) { + + "use strict"; + + if ( typeof module === "object" && typeof module.exports === "object" ) { + + // For CommonJS and CommonJS-like environments where a proper `window` + // is present, execute the factory and get jQuery. + // For environments that do not have a `window` with a `document` + // (such as Node.js), expose a factory as module.exports. + // This accentuates the need for the creation of a real `window`. + // e.g. var jQuery = require("jquery")(window); + // See ticket #14549 for more info. + module.exports = global.document ? + factory( global, true ) : + function( w ) { + if ( !w.document ) { + throw new Error( "jQuery requires a window with a document" ); + } + return factory( w ); + }; + } else { + factory( global ); + } + +// Pass this if window is not defined yet +} )( typeof window !== "undefined" ? window : this, function( window, noGlobal ) { + +// Edge <= 12 - 13+, Firefox <=18 - 45+, IE 10 - 11, Safari 5.1 - 9+, iOS 6 - 9.1 +// throw exceptions when non-strict code (e.g., ASP.NET 4.5) accesses strict mode +// arguments.callee.caller (trac-13335). But as of jQuery 3.0 (2016), strict mode should be common +// enough that all such attempts are guarded in a try block. +"use strict"; + +var arr = []; + +var getProto = Object.getPrototypeOf; + +var slice = arr.slice; + +var flat = arr.flat ? function( array ) { + return arr.flat.call( array ); +} : function( array ) { + return arr.concat.apply( [], array ); +}; + + +var push = arr.push; + +var indexOf = arr.indexOf; + +var class2type = {}; + +var toString = class2type.toString; + +var hasOwn = class2type.hasOwnProperty; + +var fnToString = hasOwn.toString; + +var ObjectFunctionString = fnToString.call( Object ); + +var support = {}; + +var isFunction = function isFunction( obj ) { + + // Support: Chrome <=57, Firefox <=52 + // In some browsers, typeof returns "function" for HTML elements + // (i.e., `typeof document.createElement( "object" ) === "function"`). + // We don't want to classify *any* DOM node as a function. + return typeof obj === "function" && typeof obj.nodeType !== "number"; + }; + + +var isWindow = function isWindow( obj ) { + return obj != null && obj === obj.window; + }; + + +var document = window.document; + + + + var preservedScriptAttributes = { + type: true, + src: true, + nonce: true, + noModule: true + }; + + function DOMEval( code, node, doc ) { + doc = doc || document; + + var i, val, + script = doc.createElement( "script" ); + + script.text = code; + if ( node ) { + for ( i in preservedScriptAttributes ) { + + // Support: Firefox 64+, Edge 18+ + // Some browsers don't support the "nonce" property on scripts. + // On the other hand, just using `getAttribute` is not enough as + // the `nonce` attribute is reset to an empty string whenever it + // becomes browsing-context connected. + // See https://github.com/whatwg/html/issues/2369 + // See https://html.spec.whatwg.org/#nonce-attributes + // The `node.getAttribute` check was added for the sake of + // `jQuery.globalEval` so that it can fake a nonce-containing node + // via an object. + val = node[ i ] || node.getAttribute && node.getAttribute( i ); + if ( val ) { + script.setAttribute( i, val ); + } + } + } + doc.head.appendChild( script ).parentNode.removeChild( script ); + } + + +function toType( obj ) { + if ( obj == null ) { + return obj + ""; + } + + // Support: Android <=2.3 only (functionish RegExp) + return typeof obj === "object" || typeof obj === "function" ? + class2type[ toString.call( obj ) ] || "object" : + typeof obj; +} +/* global Symbol */ +// Defining this global in .eslintrc.json would create a danger of using the global +// unguarded in another place, it seems safer to define global only for this module + + + +var + version = "3.5.1", + + // Define a local copy of jQuery + jQuery = function( selector, context ) { + + // The jQuery object is actually just the init constructor 'enhanced' + // Need init if jQuery is called (just allow error to be thrown if not included) + return new jQuery.fn.init( selector, context ); + }; + +jQuery.fn = jQuery.prototype = { + + // The current version of jQuery being used + jquery: version, + + constructor: jQuery, + + // The default length of a jQuery object is 0 + length: 0, + + toArray: function() { + return slice.call( this ); + }, + + // Get the Nth element in the matched element set OR + // Get the whole matched element set as a clean array + get: function( num ) { + + // Return all the elements in a clean array + if ( num == null ) { + return slice.call( this ); + } + + // Return just the one element from the set + return num < 0 ? this[ num + this.length ] : this[ num ]; + }, + + // Take an array of elements and push it onto the stack + // (returning the new matched element set) + pushStack: function( elems ) { + + // Build a new jQuery matched element set + var ret = jQuery.merge( this.constructor(), elems ); + + // Add the old object onto the stack (as a reference) + ret.prevObject = this; + + // Return the newly-formed element set + return ret; + }, + + // Execute a callback for every element in the matched set. + each: function( callback ) { + return jQuery.each( this, callback ); + }, + + map: function( callback ) { + return this.pushStack( jQuery.map( this, function( elem, i ) { + return callback.call( elem, i, elem ); + } ) ); + }, + + slice: function() { + return this.pushStack( slice.apply( this, arguments ) ); + }, + + first: function() { + return this.eq( 0 ); + }, + + last: function() { + return this.eq( -1 ); + }, + + even: function() { + return this.pushStack( jQuery.grep( this, function( _elem, i ) { + return ( i + 1 ) % 2; + } ) ); + }, + + odd: function() { + return this.pushStack( jQuery.grep( this, function( _elem, i ) { + return i % 2; + } ) ); + }, + + eq: function( i ) { + var len = this.length, + j = +i + ( i < 0 ? len : 0 ); + return this.pushStack( j >= 0 && j < len ? [ this[ j ] ] : [] ); + }, + + end: function() { + return this.prevObject || this.constructor(); + }, + + // For internal use only. + // Behaves like an Array's method, not like a jQuery method. + push: push, + sort: arr.sort, + splice: arr.splice +}; + +jQuery.extend = jQuery.fn.extend = function() { + var options, name, src, copy, copyIsArray, clone, + target = arguments[ 0 ] || {}, + i = 1, + length = arguments.length, + deep = false; + + // Handle a deep copy situation + if ( typeof target === "boolean" ) { + deep = target; + + // Skip the boolean and the target + target = arguments[ i ] || {}; + i++; + } + + // Handle case when target is a string or something (possible in deep copy) + if ( typeof target !== "object" && !isFunction( target ) ) { + target = {}; + } + + // Extend jQuery itself if only one argument is passed + if ( i === length ) { + target = this; + i--; + } + + for ( ; i < length; i++ ) { + + // Only deal with non-null/undefined values + if ( ( options = arguments[ i ] ) != null ) { + + // Extend the base object + for ( name in options ) { + copy = options[ name ]; + + // Prevent Object.prototype pollution + // Prevent never-ending loop + if ( name === "__proto__" || target === copy ) { + continue; + } + + // Recurse if we're merging plain objects or arrays + if ( deep && copy && ( jQuery.isPlainObject( copy ) || + ( copyIsArray = Array.isArray( copy ) ) ) ) { + src = target[ name ]; + + // Ensure proper type for the source value + if ( copyIsArray && !Array.isArray( src ) ) { + clone = []; + } else if ( !copyIsArray && !jQuery.isPlainObject( src ) ) { + clone = {}; + } else { + clone = src; + } + copyIsArray = false; + + // Never move original objects, clone them + target[ name ] = jQuery.extend( deep, clone, copy ); + + // Don't bring in undefined values + } else if ( copy !== undefined ) { + target[ name ] = copy; + } + } + } + } + + // Return the modified object + return target; +}; + +jQuery.extend( { + + // Unique for each copy of jQuery on the page + expando: "jQuery" + ( version + Math.random() ).replace( /\D/g, "" ), + + // Assume jQuery is ready without the ready module + isReady: true, + + error: function( msg ) { + throw new Error( msg ); + }, + + noop: function() {}, + + isPlainObject: function( obj ) { + var proto, Ctor; + + // Detect obvious negatives + // Use toString instead of jQuery.type to catch host objects + if ( !obj || toString.call( obj ) !== "[object Object]" ) { + return false; + } + + proto = getProto( obj ); + + // Objects with no prototype (e.g., `Object.create( null )`) are plain + if ( !proto ) { + return true; + } + + // Objects with prototype are plain iff they were constructed by a global Object function + Ctor = hasOwn.call( proto, "constructor" ) && proto.constructor; + return typeof Ctor === "function" && fnToString.call( Ctor ) === ObjectFunctionString; + }, + + isEmptyObject: function( obj ) { + var name; + + for ( name in obj ) { + return false; + } + return true; + }, + + // Evaluates a script in a provided context; falls back to the global one + // if not specified. + globalEval: function( code, options, doc ) { + DOMEval( code, { nonce: options && options.nonce }, doc ); + }, + + each: function( obj, callback ) { + var length, i = 0; + + if ( isArrayLike( obj ) ) { + length = obj.length; + for ( ; i < length; i++ ) { + if ( callback.call( obj[ i ], i, obj[ i ] ) === false ) { + break; + } + } + } else { + for ( i in obj ) { + if ( callback.call( obj[ i ], i, obj[ i ] ) === false ) { + break; + } + } + } + + return obj; + }, + + // results is for internal usage only + makeArray: function( arr, results ) { + var ret = results || []; + + if ( arr != null ) { + if ( isArrayLike( Object( arr ) ) ) { + jQuery.merge( ret, + typeof arr === "string" ? + [ arr ] : arr + ); + } else { + push.call( ret, arr ); + } + } + + return ret; + }, + + inArray: function( elem, arr, i ) { + return arr == null ? -1 : indexOf.call( arr, elem, i ); + }, + + // Support: Android <=4.0 only, PhantomJS 1 only + // push.apply(_, arraylike) throws on ancient WebKit + merge: function( first, second ) { + var len = +second.length, + j = 0, + i = first.length; + + for ( ; j < len; j++ ) { + first[ i++ ] = second[ j ]; + } + + first.length = i; + + return first; + }, + + grep: function( elems, callback, invert ) { + var callbackInverse, + matches = [], + i = 0, + length = elems.length, + callbackExpect = !invert; + + // Go through the array, only saving the items + // that pass the validator function + for ( ; i < length; i++ ) { + callbackInverse = !callback( elems[ i ], i ); + if ( callbackInverse !== callbackExpect ) { + matches.push( elems[ i ] ); + } + } + + return matches; + }, + + // arg is for internal usage only + map: function( elems, callback, arg ) { + var length, value, + i = 0, + ret = []; + + // Go through the array, translating each of the items to their new values + if ( isArrayLike( elems ) ) { + length = elems.length; + for ( ; i < length; i++ ) { + value = callback( elems[ i ], i, arg ); + + if ( value != null ) { + ret.push( value ); + } + } + + // Go through every key on the object, + } else { + for ( i in elems ) { + value = callback( elems[ i ], i, arg ); + + if ( value != null ) { + ret.push( value ); + } + } + } + + // Flatten any nested arrays + return flat( ret ); + }, + + // A global GUID counter for objects + guid: 1, + + // jQuery.support is not used in Core but other projects attach their + // properties to it so it needs to exist. + support: support +} ); + +if ( typeof Symbol === "function" ) { + jQuery.fn[ Symbol.iterator ] = arr[ Symbol.iterator ]; +} + +// Populate the class2type map +jQuery.each( "Boolean Number String Function Array Date RegExp Object Error Symbol".split( " " ), +function( _i, name ) { + class2type[ "[object " + name + "]" ] = name.toLowerCase(); +} ); + +function isArrayLike( obj ) { + + // Support: real iOS 8.2 only (not reproducible in simulator) + // `in` check used to prevent JIT error (gh-2145) + // hasOwn isn't used here due to false negatives + // regarding Nodelist length in IE + var length = !!obj && "length" in obj && obj.length, + type = toType( obj ); + + if ( isFunction( obj ) || isWindow( obj ) ) { + return false; + } + + return type === "array" || length === 0 || + typeof length === "number" && length > 0 && ( length - 1 ) in obj; +} +var Sizzle = +/*! + * Sizzle CSS Selector Engine v2.3.5 + * https://sizzlejs.com/ + * + * Copyright JS Foundation and other contributors + * Released under the MIT license + * https://js.foundation/ + * + * Date: 2020-03-14 + */ +( function( window ) { +var i, + support, + Expr, + getText, + isXML, + tokenize, + compile, + select, + outermostContext, + sortInput, + hasDuplicate, + + // Local document vars + setDocument, + document, + docElem, + documentIsHTML, + rbuggyQSA, + rbuggyMatches, + matches, + contains, + + // Instance-specific data + expando = "sizzle" + 1 * new Date(), + preferredDoc = window.document, + dirruns = 0, + done = 0, + classCache = createCache(), + tokenCache = createCache(), + compilerCache = createCache(), + nonnativeSelectorCache = createCache(), + sortOrder = function( a, b ) { + if ( a === b ) { + hasDuplicate = true; + } + return 0; + }, + + // Instance methods + hasOwn = ( {} ).hasOwnProperty, + arr = [], + pop = arr.pop, + pushNative = arr.push, + push = arr.push, + slice = arr.slice, + + // Use a stripped-down indexOf as it's faster than native + // https://jsperf.com/thor-indexof-vs-for/5 + indexOf = function( list, elem ) { + var i = 0, + len = list.length; + for ( ; i < len; i++ ) { + if ( list[ i ] === elem ) { + return i; + } + } + return -1; + }, + + booleans = "checked|selected|async|autofocus|autoplay|controls|defer|disabled|hidden|" + + "ismap|loop|multiple|open|readonly|required|scoped", + + // Regular expressions + + // http://www.w3.org/TR/css3-selectors/#whitespace + whitespace = "[\\x20\\t\\r\\n\\f]", + + // https://www.w3.org/TR/css-syntax-3/#ident-token-diagram + identifier = "(?:\\\\[\\da-fA-F]{1,6}" + whitespace + + "?|\\\\[^\\r\\n\\f]|[\\w-]|[^\0-\\x7f])+", + + // Attribute selectors: http://www.w3.org/TR/selectors/#attribute-selectors + attributes = "\\[" + whitespace + "*(" + identifier + ")(?:" + whitespace + + + // Operator (capture 2) + "*([*^$|!~]?=)" + whitespace + + + // "Attribute values must be CSS identifiers [capture 5] + // or strings [capture 3 or capture 4]" + "*(?:'((?:\\\\.|[^\\\\'])*)'|\"((?:\\\\.|[^\\\\\"])*)\"|(" + identifier + "))|)" + + whitespace + "*\\]", + + pseudos = ":(" + identifier + ")(?:\\((" + + + // To reduce the number of selectors needing tokenize in the preFilter, prefer arguments: + // 1. quoted (capture 3; capture 4 or capture 5) + "('((?:\\\\.|[^\\\\'])*)'|\"((?:\\\\.|[^\\\\\"])*)\")|" + + + // 2. simple (capture 6) + "((?:\\\\.|[^\\\\()[\\]]|" + attributes + ")*)|" + + + // 3. anything else (capture 2) + ".*" + + ")\\)|)", + + // Leading and non-escaped trailing whitespace, capturing some non-whitespace characters preceding the latter + rwhitespace = new RegExp( whitespace + "+", "g" ), + rtrim = new RegExp( "^" + whitespace + "+|((?:^|[^\\\\])(?:\\\\.)*)" + + whitespace + "+$", "g" ), + + rcomma = new RegExp( "^" + whitespace + "*," + whitespace + "*" ), + rcombinators = new RegExp( "^" + whitespace + "*([>+~]|" + whitespace + ")" + whitespace + + "*" ), + rdescend = new RegExp( whitespace + "|>" ), + + rpseudo = new RegExp( pseudos ), + ridentifier = new RegExp( "^" + identifier + "$" ), + + matchExpr = { + "ID": new RegExp( "^#(" + identifier + ")" ), + "CLASS": new RegExp( "^\\.(" + identifier + ")" ), + "TAG": new RegExp( "^(" + identifier + "|[*])" ), + "ATTR": new RegExp( "^" + attributes ), + "PSEUDO": new RegExp( "^" + pseudos ), + "CHILD": new RegExp( "^:(only|first|last|nth|nth-last)-(child|of-type)(?:\\(" + + whitespace + "*(even|odd|(([+-]|)(\\d*)n|)" + whitespace + "*(?:([+-]|)" + + whitespace + "*(\\d+)|))" + whitespace + "*\\)|)", "i" ), + "bool": new RegExp( "^(?:" + booleans + ")$", "i" ), + + // For use in libraries implementing .is() + // We use this for POS matching in `select` + "needsContext": new RegExp( "^" + whitespace + + "*[>+~]|:(even|odd|eq|gt|lt|nth|first|last)(?:\\(" + whitespace + + "*((?:-\\d)?\\d*)" + whitespace + "*\\)|)(?=[^-]|$)", "i" ) + }, + + rhtml = /HTML$/i, + rinputs = /^(?:input|select|textarea|button)$/i, + rheader = /^h\d$/i, + + rnative = /^[^{]+\{\s*\[native \w/, + + // Easily-parseable/retrievable ID or TAG or CLASS selectors + rquickExpr = /^(?:#([\w-]+)|(\w+)|\.([\w-]+))$/, + + rsibling = /[+~]/, + + // CSS escapes + // http://www.w3.org/TR/CSS21/syndata.html#escaped-characters + runescape = new RegExp( "\\\\[\\da-fA-F]{1,6}" + whitespace + "?|\\\\([^\\r\\n\\f])", "g" ), + funescape = function( escape, nonHex ) { + var high = "0x" + escape.slice( 1 ) - 0x10000; + + return nonHex ? + + // Strip the backslash prefix from a non-hex escape sequence + nonHex : + + // Replace a hexadecimal escape sequence with the encoded Unicode code point + // Support: IE <=11+ + // For values outside the Basic Multilingual Plane (BMP), manually construct a + // surrogate pair + high < 0 ? + String.fromCharCode( high + 0x10000 ) : + String.fromCharCode( high >> 10 | 0xD800, high & 0x3FF | 0xDC00 ); + }, + + // CSS string/identifier serialization + // https://drafts.csswg.org/cssom/#common-serializing-idioms + rcssescape = /([\0-\x1f\x7f]|^-?\d)|^-$|[^\0-\x1f\x7f-\uFFFF\w-]/g, + fcssescape = function( ch, asCodePoint ) { + if ( asCodePoint ) { + + // U+0000 NULL becomes U+FFFD REPLACEMENT CHARACTER + if ( ch === "\0" ) { + return "\uFFFD"; + } + + // Control characters and (dependent upon position) numbers get escaped as code points + return ch.slice( 0, -1 ) + "\\" + + ch.charCodeAt( ch.length - 1 ).toString( 16 ) + " "; + } + + // Other potentially-special ASCII characters get backslash-escaped + return "\\" + ch; + }, + + // Used for iframes + // See setDocument() + // Removing the function wrapper causes a "Permission Denied" + // error in IE + unloadHandler = function() { + setDocument(); + }, + + inDisabledFieldset = addCombinator( + function( elem ) { + return elem.disabled === true && elem.nodeName.toLowerCase() === "fieldset"; + }, + { dir: "parentNode", next: "legend" } + ); + +// Optimize for push.apply( _, NodeList ) +try { + push.apply( + ( arr = slice.call( preferredDoc.childNodes ) ), + preferredDoc.childNodes + ); + + // Support: Android<4.0 + // Detect silently failing push.apply + // eslint-disable-next-line no-unused-expressions + arr[ preferredDoc.childNodes.length ].nodeType; +} catch ( e ) { + push = { apply: arr.length ? + + // Leverage slice if possible + function( target, els ) { + pushNative.apply( target, slice.call( els ) ); + } : + + // Support: IE<9 + // Otherwise append directly + function( target, els ) { + var j = target.length, + i = 0; + + // Can't trust NodeList.length + while ( ( target[ j++ ] = els[ i++ ] ) ) {} + target.length = j - 1; + } + }; +} + +function Sizzle( selector, context, results, seed ) { + var m, i, elem, nid, match, groups, newSelector, + newContext = context && context.ownerDocument, + + // nodeType defaults to 9, since context defaults to document + nodeType = context ? context.nodeType : 9; + + results = results || []; + + // Return early from calls with invalid selector or context + if ( typeof selector !== "string" || !selector || + nodeType !== 1 && nodeType !== 9 && nodeType !== 11 ) { + + return results; + } + + // Try to shortcut find operations (as opposed to filters) in HTML documents + if ( !seed ) { + setDocument( context ); + context = context || document; + + if ( documentIsHTML ) { + + // If the selector is sufficiently simple, try using a "get*By*" DOM method + // (excepting DocumentFragment context, where the methods don't exist) + if ( nodeType !== 11 && ( match = rquickExpr.exec( selector ) ) ) { + + // ID selector + if ( ( m = match[ 1 ] ) ) { + + // Document context + if ( nodeType === 9 ) { + if ( ( elem = context.getElementById( m ) ) ) { + + // Support: IE, Opera, Webkit + // TODO: identify versions + // getElementById can match elements by name instead of ID + if ( elem.id === m ) { + results.push( elem ); + return results; + } + } else { + return results; + } + + // Element context + } else { + + // Support: IE, Opera, Webkit + // TODO: identify versions + // getElementById can match elements by name instead of ID + if ( newContext && ( elem = newContext.getElementById( m ) ) && + contains( context, elem ) && + elem.id === m ) { + + results.push( elem ); + return results; + } + } + + // Type selector + } else if ( match[ 2 ] ) { + push.apply( results, context.getElementsByTagName( selector ) ); + return results; + + // Class selector + } else if ( ( m = match[ 3 ] ) && support.getElementsByClassName && + context.getElementsByClassName ) { + + push.apply( results, context.getElementsByClassName( m ) ); + return results; + } + } + + // Take advantage of querySelectorAll + if ( support.qsa && + !nonnativeSelectorCache[ selector + " " ] && + ( !rbuggyQSA || !rbuggyQSA.test( selector ) ) && + + // Support: IE 8 only + // Exclude object elements + ( nodeType !== 1 || context.nodeName.toLowerCase() !== "object" ) ) { + + newSelector = selector; + newContext = context; + + // qSA considers elements outside a scoping root when evaluating child or + // descendant combinators, which is not what we want. + // In such cases, we work around the behavior by prefixing every selector in the + // list with an ID selector referencing the scope context. + // The technique has to be used as well when a leading combinator is used + // as such selectors are not recognized by querySelectorAll. + // Thanks to Andrew Dupont for this technique. + if ( nodeType === 1 && + ( rdescend.test( selector ) || rcombinators.test( selector ) ) ) { + + // Expand context for sibling selectors + newContext = rsibling.test( selector ) && testContext( context.parentNode ) || + context; + + // We can use :scope instead of the ID hack if the browser + // supports it & if we're not changing the context. + if ( newContext !== context || !support.scope ) { + + // Capture the context ID, setting it first if necessary + if ( ( nid = context.getAttribute( "id" ) ) ) { + nid = nid.replace( rcssescape, fcssescape ); + } else { + context.setAttribute( "id", ( nid = expando ) ); + } + } + + // Prefix every selector in the list + groups = tokenize( selector ); + i = groups.length; + while ( i-- ) { + groups[ i ] = ( nid ? "#" + nid : ":scope" ) + " " + + toSelector( groups[ i ] ); + } + newSelector = groups.join( "," ); + } + + try { + push.apply( results, + newContext.querySelectorAll( newSelector ) + ); + return results; + } catch ( qsaError ) { + nonnativeSelectorCache( selector, true ); + } finally { + if ( nid === expando ) { + context.removeAttribute( "id" ); + } + } + } + } + } + + // All others + return select( selector.replace( rtrim, "$1" ), context, results, seed ); +} + +/** + * Create key-value caches of limited size + * @returns {function(string, object)} Returns the Object data after storing it on itself with + * property name the (space-suffixed) string and (if the cache is larger than Expr.cacheLength) + * deleting the oldest entry + */ +function createCache() { + var keys = []; + + function cache( key, value ) { + + // Use (key + " ") to avoid collision with native prototype properties (see Issue #157) + if ( keys.push( key + " " ) > Expr.cacheLength ) { + + // Only keep the most recent entries + delete cache[ keys.shift() ]; + } + return ( cache[ key + " " ] = value ); + } + return cache; +} + +/** + * Mark a function for special use by Sizzle + * @param {Function} fn The function to mark + */ +function markFunction( fn ) { + fn[ expando ] = true; + return fn; +} + +/** + * Support testing using an element + * @param {Function} fn Passed the created element and returns a boolean result + */ +function assert( fn ) { + var el = document.createElement( "fieldset" ); + + try { + return !!fn( el ); + } catch ( e ) { + return false; + } finally { + + // Remove from its parent by default + if ( el.parentNode ) { + el.parentNode.removeChild( el ); + } + + // release memory in IE + el = null; + } +} + +/** + * Adds the same handler for all of the specified attrs + * @param {String} attrs Pipe-separated list of attributes + * @param {Function} handler The method that will be applied + */ +function addHandle( attrs, handler ) { + var arr = attrs.split( "|" ), + i = arr.length; + + while ( i-- ) { + Expr.attrHandle[ arr[ i ] ] = handler; + } +} + +/** + * Checks document order of two siblings + * @param {Element} a + * @param {Element} b + * @returns {Number} Returns less than 0 if a precedes b, greater than 0 if a follows b + */ +function siblingCheck( a, b ) { + var cur = b && a, + diff = cur && a.nodeType === 1 && b.nodeType === 1 && + a.sourceIndex - b.sourceIndex; + + // Use IE sourceIndex if available on both nodes + if ( diff ) { + return diff; + } + + // Check if b follows a + if ( cur ) { + while ( ( cur = cur.nextSibling ) ) { + if ( cur === b ) { + return -1; + } + } + } + + return a ? 1 : -1; +} + +/** + * Returns a function to use in pseudos for input types + * @param {String} type + */ +function createInputPseudo( type ) { + return function( elem ) { + var name = elem.nodeName.toLowerCase(); + return name === "input" && elem.type === type; + }; +} + +/** + * Returns a function to use in pseudos for buttons + * @param {String} type + */ +function createButtonPseudo( type ) { + return function( elem ) { + var name = elem.nodeName.toLowerCase(); + return ( name === "input" || name === "button" ) && elem.type === type; + }; +} + +/** + * Returns a function to use in pseudos for :enabled/:disabled + * @param {Boolean} disabled true for :disabled; false for :enabled + */ +function createDisabledPseudo( disabled ) { + + // Known :disabled false positives: fieldset[disabled] > legend:nth-of-type(n+2) :can-disable + return function( elem ) { + + // Only certain elements can match :enabled or :disabled + // https://html.spec.whatwg.org/multipage/scripting.html#selector-enabled + // https://html.spec.whatwg.org/multipage/scripting.html#selector-disabled + if ( "form" in elem ) { + + // Check for inherited disabledness on relevant non-disabled elements: + // * listed form-associated elements in a disabled fieldset + // https://html.spec.whatwg.org/multipage/forms.html#category-listed + // https://html.spec.whatwg.org/multipage/forms.html#concept-fe-disabled + // * option elements in a disabled optgroup + // https://html.spec.whatwg.org/multipage/forms.html#concept-option-disabled + // All such elements have a "form" property. + if ( elem.parentNode && elem.disabled === false ) { + + // Option elements defer to a parent optgroup if present + if ( "label" in elem ) { + if ( "label" in elem.parentNode ) { + return elem.parentNode.disabled === disabled; + } else { + return elem.disabled === disabled; + } + } + + // Support: IE 6 - 11 + // Use the isDisabled shortcut property to check for disabled fieldset ancestors + return elem.isDisabled === disabled || + + // Where there is no isDisabled, check manually + /* jshint -W018 */ + elem.isDisabled !== !disabled && + inDisabledFieldset( elem ) === disabled; + } + + return elem.disabled === disabled; + + // Try to winnow out elements that can't be disabled before trusting the disabled property. + // Some victims get caught in our net (label, legend, menu, track), but it shouldn't + // even exist on them, let alone have a boolean value. + } else if ( "label" in elem ) { + return elem.disabled === disabled; + } + + // Remaining elements are neither :enabled nor :disabled + return false; + }; +} + +/** + * Returns a function to use in pseudos for positionals + * @param {Function} fn + */ +function createPositionalPseudo( fn ) { + return markFunction( function( argument ) { + argument = +argument; + return markFunction( function( seed, matches ) { + var j, + matchIndexes = fn( [], seed.length, argument ), + i = matchIndexes.length; + + // Match elements found at the specified indexes + while ( i-- ) { + if ( seed[ ( j = matchIndexes[ i ] ) ] ) { + seed[ j ] = !( matches[ j ] = seed[ j ] ); + } + } + } ); + } ); +} + +/** + * Checks a node for validity as a Sizzle context + * @param {Element|Object=} context + * @returns {Element|Object|Boolean} The input node if acceptable, otherwise a falsy value + */ +function testContext( context ) { + return context && typeof context.getElementsByTagName !== "undefined" && context; +} + +// Expose support vars for convenience +support = Sizzle.support = {}; + +/** + * Detects XML nodes + * @param {Element|Object} elem An element or a document + * @returns {Boolean} True iff elem is a non-HTML XML node + */ +isXML = Sizzle.isXML = function( elem ) { + var namespace = elem.namespaceURI, + docElem = ( elem.ownerDocument || elem ).documentElement; + + // Support: IE <=8 + // Assume HTML when documentElement doesn't yet exist, such as inside loading iframes + // https://bugs.jquery.com/ticket/4833 + return !rhtml.test( namespace || docElem && docElem.nodeName || "HTML" ); +}; + +/** + * Sets document-related variables once based on the current document + * @param {Element|Object} [doc] An element or document object to use to set the document + * @returns {Object} Returns the current document + */ +setDocument = Sizzle.setDocument = function( node ) { + var hasCompare, subWindow, + doc = node ? node.ownerDocument || node : preferredDoc; + + // Return early if doc is invalid or already selected + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + // eslint-disable-next-line eqeqeq + if ( doc == document || doc.nodeType !== 9 || !doc.documentElement ) { + return document; + } + + // Update global variables + document = doc; + docElem = document.documentElement; + documentIsHTML = !isXML( document ); + + // Support: IE 9 - 11+, Edge 12 - 18+ + // Accessing iframe documents after unload throws "permission denied" errors (jQuery #13936) + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + // eslint-disable-next-line eqeqeq + if ( preferredDoc != document && + ( subWindow = document.defaultView ) && subWindow.top !== subWindow ) { + + // Support: IE 11, Edge + if ( subWindow.addEventListener ) { + subWindow.addEventListener( "unload", unloadHandler, false ); + + // Support: IE 9 - 10 only + } else if ( subWindow.attachEvent ) { + subWindow.attachEvent( "onunload", unloadHandler ); + } + } + + // Support: IE 8 - 11+, Edge 12 - 18+, Chrome <=16 - 25 only, Firefox <=3.6 - 31 only, + // Safari 4 - 5 only, Opera <=11.6 - 12.x only + // IE/Edge & older browsers don't support the :scope pseudo-class. + // Support: Safari 6.0 only + // Safari 6.0 supports :scope but it's an alias of :root there. + support.scope = assert( function( el ) { + docElem.appendChild( el ).appendChild( document.createElement( "div" ) ); + return typeof el.querySelectorAll !== "undefined" && + !el.querySelectorAll( ":scope fieldset div" ).length; + } ); + + /* Attributes + ---------------------------------------------------------------------- */ + + // Support: IE<8 + // Verify that getAttribute really returns attributes and not properties + // (excepting IE8 booleans) + support.attributes = assert( function( el ) { + el.className = "i"; + return !el.getAttribute( "className" ); + } ); + + /* getElement(s)By* + ---------------------------------------------------------------------- */ + + // Check if getElementsByTagName("*") returns only elements + support.getElementsByTagName = assert( function( el ) { + el.appendChild( document.createComment( "" ) ); + return !el.getElementsByTagName( "*" ).length; + } ); + + // Support: IE<9 + support.getElementsByClassName = rnative.test( document.getElementsByClassName ); + + // Support: IE<10 + // Check if getElementById returns elements by name + // The broken getElementById methods don't pick up programmatically-set names, + // so use a roundabout getElementsByName test + support.getById = assert( function( el ) { + docElem.appendChild( el ).id = expando; + return !document.getElementsByName || !document.getElementsByName( expando ).length; + } ); + + // ID filter and find + if ( support.getById ) { + Expr.filter[ "ID" ] = function( id ) { + var attrId = id.replace( runescape, funescape ); + return function( elem ) { + return elem.getAttribute( "id" ) === attrId; + }; + }; + Expr.find[ "ID" ] = function( id, context ) { + if ( typeof context.getElementById !== "undefined" && documentIsHTML ) { + var elem = context.getElementById( id ); + return elem ? [ elem ] : []; + } + }; + } else { + Expr.filter[ "ID" ] = function( id ) { + var attrId = id.replace( runescape, funescape ); + return function( elem ) { + var node = typeof elem.getAttributeNode !== "undefined" && + elem.getAttributeNode( "id" ); + return node && node.value === attrId; + }; + }; + + // Support: IE 6 - 7 only + // getElementById is not reliable as a find shortcut + Expr.find[ "ID" ] = function( id, context ) { + if ( typeof context.getElementById !== "undefined" && documentIsHTML ) { + var node, i, elems, + elem = context.getElementById( id ); + + if ( elem ) { + + // Verify the id attribute + node = elem.getAttributeNode( "id" ); + if ( node && node.value === id ) { + return [ elem ]; + } + + // Fall back on getElementsByName + elems = context.getElementsByName( id ); + i = 0; + while ( ( elem = elems[ i++ ] ) ) { + node = elem.getAttributeNode( "id" ); + if ( node && node.value === id ) { + return [ elem ]; + } + } + } + + return []; + } + }; + } + + // Tag + Expr.find[ "TAG" ] = support.getElementsByTagName ? + function( tag, context ) { + if ( typeof context.getElementsByTagName !== "undefined" ) { + return context.getElementsByTagName( tag ); + + // DocumentFragment nodes don't have gEBTN + } else if ( support.qsa ) { + return context.querySelectorAll( tag ); + } + } : + + function( tag, context ) { + var elem, + tmp = [], + i = 0, + + // By happy coincidence, a (broken) gEBTN appears on DocumentFragment nodes too + results = context.getElementsByTagName( tag ); + + // Filter out possible comments + if ( tag === "*" ) { + while ( ( elem = results[ i++ ] ) ) { + if ( elem.nodeType === 1 ) { + tmp.push( elem ); + } + } + + return tmp; + } + return results; + }; + + // Class + Expr.find[ "CLASS" ] = support.getElementsByClassName && function( className, context ) { + if ( typeof context.getElementsByClassName !== "undefined" && documentIsHTML ) { + return context.getElementsByClassName( className ); + } + }; + + /* QSA/matchesSelector + ---------------------------------------------------------------------- */ + + // QSA and matchesSelector support + + // matchesSelector(:active) reports false when true (IE9/Opera 11.5) + rbuggyMatches = []; + + // qSa(:focus) reports false when true (Chrome 21) + // We allow this because of a bug in IE8/9 that throws an error + // whenever `document.activeElement` is accessed on an iframe + // So, we allow :focus to pass through QSA all the time to avoid the IE error + // See https://bugs.jquery.com/ticket/13378 + rbuggyQSA = []; + + if ( ( support.qsa = rnative.test( document.querySelectorAll ) ) ) { + + // Build QSA regex + // Regex strategy adopted from Diego Perini + assert( function( el ) { + + var input; + + // Select is set to empty string on purpose + // This is to test IE's treatment of not explicitly + // setting a boolean content attribute, + // since its presence should be enough + // https://bugs.jquery.com/ticket/12359 + docElem.appendChild( el ).innerHTML = "" + + ""; + + // Support: IE8, Opera 11-12.16 + // Nothing should be selected when empty strings follow ^= or $= or *= + // The test attribute must be unknown in Opera but "safe" for WinRT + // https://msdn.microsoft.com/en-us/library/ie/hh465388.aspx#attribute_section + if ( el.querySelectorAll( "[msallowcapture^='']" ).length ) { + rbuggyQSA.push( "[*^$]=" + whitespace + "*(?:''|\"\")" ); + } + + // Support: IE8 + // Boolean attributes and "value" are not treated correctly + if ( !el.querySelectorAll( "[selected]" ).length ) { + rbuggyQSA.push( "\\[" + whitespace + "*(?:value|" + booleans + ")" ); + } + + // Support: Chrome<29, Android<4.4, Safari<7.0+, iOS<7.0+, PhantomJS<1.9.8+ + if ( !el.querySelectorAll( "[id~=" + expando + "-]" ).length ) { + rbuggyQSA.push( "~=" ); + } + + // Support: IE 11+, Edge 15 - 18+ + // IE 11/Edge don't find elements on a `[name='']` query in some cases. + // Adding a temporary attribute to the document before the selection works + // around the issue. + // Interestingly, IE 10 & older don't seem to have the issue. + input = document.createElement( "input" ); + input.setAttribute( "name", "" ); + el.appendChild( input ); + if ( !el.querySelectorAll( "[name='']" ).length ) { + rbuggyQSA.push( "\\[" + whitespace + "*name" + whitespace + "*=" + + whitespace + "*(?:''|\"\")" ); + } + + // Webkit/Opera - :checked should return selected option elements + // http://www.w3.org/TR/2011/REC-css3-selectors-20110929/#checked + // IE8 throws error here and will not see later tests + if ( !el.querySelectorAll( ":checked" ).length ) { + rbuggyQSA.push( ":checked" ); + } + + // Support: Safari 8+, iOS 8+ + // https://bugs.webkit.org/show_bug.cgi?id=136851 + // In-page `selector#id sibling-combinator selector` fails + if ( !el.querySelectorAll( "a#" + expando + "+*" ).length ) { + rbuggyQSA.push( ".#.+[+~]" ); + } + + // Support: Firefox <=3.6 - 5 only + // Old Firefox doesn't throw on a badly-escaped identifier. + el.querySelectorAll( "\\\f" ); + rbuggyQSA.push( "[\\r\\n\\f]" ); + } ); + + assert( function( el ) { + el.innerHTML = "" + + ""; + + // Support: Windows 8 Native Apps + // The type and name attributes are restricted during .innerHTML assignment + var input = document.createElement( "input" ); + input.setAttribute( "type", "hidden" ); + el.appendChild( input ).setAttribute( "name", "D" ); + + // Support: IE8 + // Enforce case-sensitivity of name attribute + if ( el.querySelectorAll( "[name=d]" ).length ) { + rbuggyQSA.push( "name" + whitespace + "*[*^$|!~]?=" ); + } + + // FF 3.5 - :enabled/:disabled and hidden elements (hidden elements are still enabled) + // IE8 throws error here and will not see later tests + if ( el.querySelectorAll( ":enabled" ).length !== 2 ) { + rbuggyQSA.push( ":enabled", ":disabled" ); + } + + // Support: IE9-11+ + // IE's :disabled selector does not pick up the children of disabled fieldsets + docElem.appendChild( el ).disabled = true; + if ( el.querySelectorAll( ":disabled" ).length !== 2 ) { + rbuggyQSA.push( ":enabled", ":disabled" ); + } + + // Support: Opera 10 - 11 only + // Opera 10-11 does not throw on post-comma invalid pseudos + el.querySelectorAll( "*,:x" ); + rbuggyQSA.push( ",.*:" ); + } ); + } + + if ( ( support.matchesSelector = rnative.test( ( matches = docElem.matches || + docElem.webkitMatchesSelector || + docElem.mozMatchesSelector || + docElem.oMatchesSelector || + docElem.msMatchesSelector ) ) ) ) { + + assert( function( el ) { + + // Check to see if it's possible to do matchesSelector + // on a disconnected node (IE 9) + support.disconnectedMatch = matches.call( el, "*" ); + + // This should fail with an exception + // Gecko does not error, returns false instead + matches.call( el, "[s!='']:x" ); + rbuggyMatches.push( "!=", pseudos ); + } ); + } + + rbuggyQSA = rbuggyQSA.length && new RegExp( rbuggyQSA.join( "|" ) ); + rbuggyMatches = rbuggyMatches.length && new RegExp( rbuggyMatches.join( "|" ) ); + + /* Contains + ---------------------------------------------------------------------- */ + hasCompare = rnative.test( docElem.compareDocumentPosition ); + + // Element contains another + // Purposefully self-exclusive + // As in, an element does not contain itself + contains = hasCompare || rnative.test( docElem.contains ) ? + function( a, b ) { + var adown = a.nodeType === 9 ? a.documentElement : a, + bup = b && b.parentNode; + return a === bup || !!( bup && bup.nodeType === 1 && ( + adown.contains ? + adown.contains( bup ) : + a.compareDocumentPosition && a.compareDocumentPosition( bup ) & 16 + ) ); + } : + function( a, b ) { + if ( b ) { + while ( ( b = b.parentNode ) ) { + if ( b === a ) { + return true; + } + } + } + return false; + }; + + /* Sorting + ---------------------------------------------------------------------- */ + + // Document order sorting + sortOrder = hasCompare ? + function( a, b ) { + + // Flag for duplicate removal + if ( a === b ) { + hasDuplicate = true; + return 0; + } + + // Sort on method existence if only one input has compareDocumentPosition + var compare = !a.compareDocumentPosition - !b.compareDocumentPosition; + if ( compare ) { + return compare; + } + + // Calculate position if both inputs belong to the same document + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + // eslint-disable-next-line eqeqeq + compare = ( a.ownerDocument || a ) == ( b.ownerDocument || b ) ? + a.compareDocumentPosition( b ) : + + // Otherwise we know they are disconnected + 1; + + // Disconnected nodes + if ( compare & 1 || + ( !support.sortDetached && b.compareDocumentPosition( a ) === compare ) ) { + + // Choose the first element that is related to our preferred document + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + // eslint-disable-next-line eqeqeq + if ( a == document || a.ownerDocument == preferredDoc && + contains( preferredDoc, a ) ) { + return -1; + } + + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + // eslint-disable-next-line eqeqeq + if ( b == document || b.ownerDocument == preferredDoc && + contains( preferredDoc, b ) ) { + return 1; + } + + // Maintain original order + return sortInput ? + ( indexOf( sortInput, a ) - indexOf( sortInput, b ) ) : + 0; + } + + return compare & 4 ? -1 : 1; + } : + function( a, b ) { + + // Exit early if the nodes are identical + if ( a === b ) { + hasDuplicate = true; + return 0; + } + + var cur, + i = 0, + aup = a.parentNode, + bup = b.parentNode, + ap = [ a ], + bp = [ b ]; + + // Parentless nodes are either documents or disconnected + if ( !aup || !bup ) { + + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + /* eslint-disable eqeqeq */ + return a == document ? -1 : + b == document ? 1 : + /* eslint-enable eqeqeq */ + aup ? -1 : + bup ? 1 : + sortInput ? + ( indexOf( sortInput, a ) - indexOf( sortInput, b ) ) : + 0; + + // If the nodes are siblings, we can do a quick check + } else if ( aup === bup ) { + return siblingCheck( a, b ); + } + + // Otherwise we need full lists of their ancestors for comparison + cur = a; + while ( ( cur = cur.parentNode ) ) { + ap.unshift( cur ); + } + cur = b; + while ( ( cur = cur.parentNode ) ) { + bp.unshift( cur ); + } + + // Walk down the tree looking for a discrepancy + while ( ap[ i ] === bp[ i ] ) { + i++; + } + + return i ? + + // Do a sibling check if the nodes have a common ancestor + siblingCheck( ap[ i ], bp[ i ] ) : + + // Otherwise nodes in our document sort first + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + /* eslint-disable eqeqeq */ + ap[ i ] == preferredDoc ? -1 : + bp[ i ] == preferredDoc ? 1 : + /* eslint-enable eqeqeq */ + 0; + }; + + return document; +}; + +Sizzle.matches = function( expr, elements ) { + return Sizzle( expr, null, null, elements ); +}; + +Sizzle.matchesSelector = function( elem, expr ) { + setDocument( elem ); + + if ( support.matchesSelector && documentIsHTML && + !nonnativeSelectorCache[ expr + " " ] && + ( !rbuggyMatches || !rbuggyMatches.test( expr ) ) && + ( !rbuggyQSA || !rbuggyQSA.test( expr ) ) ) { + + try { + var ret = matches.call( elem, expr ); + + // IE 9's matchesSelector returns false on disconnected nodes + if ( ret || support.disconnectedMatch || + + // As well, disconnected nodes are said to be in a document + // fragment in IE 9 + elem.document && elem.document.nodeType !== 11 ) { + return ret; + } + } catch ( e ) { + nonnativeSelectorCache( expr, true ); + } + } + + return Sizzle( expr, document, null, [ elem ] ).length > 0; +}; + +Sizzle.contains = function( context, elem ) { + + // Set document vars if needed + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + // eslint-disable-next-line eqeqeq + if ( ( context.ownerDocument || context ) != document ) { + setDocument( context ); + } + return contains( context, elem ); +}; + +Sizzle.attr = function( elem, name ) { + + // Set document vars if needed + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + // eslint-disable-next-line eqeqeq + if ( ( elem.ownerDocument || elem ) != document ) { + setDocument( elem ); + } + + var fn = Expr.attrHandle[ name.toLowerCase() ], + + // Don't get fooled by Object.prototype properties (jQuery #13807) + val = fn && hasOwn.call( Expr.attrHandle, name.toLowerCase() ) ? + fn( elem, name, !documentIsHTML ) : + undefined; + + return val !== undefined ? + val : + support.attributes || !documentIsHTML ? + elem.getAttribute( name ) : + ( val = elem.getAttributeNode( name ) ) && val.specified ? + val.value : + null; +}; + +Sizzle.escape = function( sel ) { + return ( sel + "" ).replace( rcssescape, fcssescape ); +}; + +Sizzle.error = function( msg ) { + throw new Error( "Syntax error, unrecognized expression: " + msg ); +}; + +/** + * Document sorting and removing duplicates + * @param {ArrayLike} results + */ +Sizzle.uniqueSort = function( results ) { + var elem, + duplicates = [], + j = 0, + i = 0; + + // Unless we *know* we can detect duplicates, assume their presence + hasDuplicate = !support.detectDuplicates; + sortInput = !support.sortStable && results.slice( 0 ); + results.sort( sortOrder ); + + if ( hasDuplicate ) { + while ( ( elem = results[ i++ ] ) ) { + if ( elem === results[ i ] ) { + j = duplicates.push( i ); + } + } + while ( j-- ) { + results.splice( duplicates[ j ], 1 ); + } + } + + // Clear input after sorting to release objects + // See https://github.com/jquery/sizzle/pull/225 + sortInput = null; + + return results; +}; + +/** + * Utility function for retrieving the text value of an array of DOM nodes + * @param {Array|Element} elem + */ +getText = Sizzle.getText = function( elem ) { + var node, + ret = "", + i = 0, + nodeType = elem.nodeType; + + if ( !nodeType ) { + + // If no nodeType, this is expected to be an array + while ( ( node = elem[ i++ ] ) ) { + + // Do not traverse comment nodes + ret += getText( node ); + } + } else if ( nodeType === 1 || nodeType === 9 || nodeType === 11 ) { + + // Use textContent for elements + // innerText usage removed for consistency of new lines (jQuery #11153) + if ( typeof elem.textContent === "string" ) { + return elem.textContent; + } else { + + // Traverse its children + for ( elem = elem.firstChild; elem; elem = elem.nextSibling ) { + ret += getText( elem ); + } + } + } else if ( nodeType === 3 || nodeType === 4 ) { + return elem.nodeValue; + } + + // Do not include comment or processing instruction nodes + + return ret; +}; + +Expr = Sizzle.selectors = { + + // Can be adjusted by the user + cacheLength: 50, + + createPseudo: markFunction, + + match: matchExpr, + + attrHandle: {}, + + find: {}, + + relative: { + ">": { dir: "parentNode", first: true }, + " ": { dir: "parentNode" }, + "+": { dir: "previousSibling", first: true }, + "~": { dir: "previousSibling" } + }, + + preFilter: { + "ATTR": function( match ) { + match[ 1 ] = match[ 1 ].replace( runescape, funescape ); + + // Move the given value to match[3] whether quoted or unquoted + match[ 3 ] = ( match[ 3 ] || match[ 4 ] || + match[ 5 ] || "" ).replace( runescape, funescape ); + + if ( match[ 2 ] === "~=" ) { + match[ 3 ] = " " + match[ 3 ] + " "; + } + + return match.slice( 0, 4 ); + }, + + "CHILD": function( match ) { + + /* matches from matchExpr["CHILD"] + 1 type (only|nth|...) + 2 what (child|of-type) + 3 argument (even|odd|\d*|\d*n([+-]\d+)?|...) + 4 xn-component of xn+y argument ([+-]?\d*n|) + 5 sign of xn-component + 6 x of xn-component + 7 sign of y-component + 8 y of y-component + */ + match[ 1 ] = match[ 1 ].toLowerCase(); + + if ( match[ 1 ].slice( 0, 3 ) === "nth" ) { + + // nth-* requires argument + if ( !match[ 3 ] ) { + Sizzle.error( match[ 0 ] ); + } + + // numeric x and y parameters for Expr.filter.CHILD + // remember that false/true cast respectively to 0/1 + match[ 4 ] = +( match[ 4 ] ? + match[ 5 ] + ( match[ 6 ] || 1 ) : + 2 * ( match[ 3 ] === "even" || match[ 3 ] === "odd" ) ); + match[ 5 ] = +( ( match[ 7 ] + match[ 8 ] ) || match[ 3 ] === "odd" ); + + // other types prohibit arguments + } else if ( match[ 3 ] ) { + Sizzle.error( match[ 0 ] ); + } + + return match; + }, + + "PSEUDO": function( match ) { + var excess, + unquoted = !match[ 6 ] && match[ 2 ]; + + if ( matchExpr[ "CHILD" ].test( match[ 0 ] ) ) { + return null; + } + + // Accept quoted arguments as-is + if ( match[ 3 ] ) { + match[ 2 ] = match[ 4 ] || match[ 5 ] || ""; + + // Strip excess characters from unquoted arguments + } else if ( unquoted && rpseudo.test( unquoted ) && + + // Get excess from tokenize (recursively) + ( excess = tokenize( unquoted, true ) ) && + + // advance to the next closing parenthesis + ( excess = unquoted.indexOf( ")", unquoted.length - excess ) - unquoted.length ) ) { + + // excess is a negative index + match[ 0 ] = match[ 0 ].slice( 0, excess ); + match[ 2 ] = unquoted.slice( 0, excess ); + } + + // Return only captures needed by the pseudo filter method (type and argument) + return match.slice( 0, 3 ); + } + }, + + filter: { + + "TAG": function( nodeNameSelector ) { + var nodeName = nodeNameSelector.replace( runescape, funescape ).toLowerCase(); + return nodeNameSelector === "*" ? + function() { + return true; + } : + function( elem ) { + return elem.nodeName && elem.nodeName.toLowerCase() === nodeName; + }; + }, + + "CLASS": function( className ) { + var pattern = classCache[ className + " " ]; + + return pattern || + ( pattern = new RegExp( "(^|" + whitespace + + ")" + className + "(" + whitespace + "|$)" ) ) && classCache( + className, function( elem ) { + return pattern.test( + typeof elem.className === "string" && elem.className || + typeof elem.getAttribute !== "undefined" && + elem.getAttribute( "class" ) || + "" + ); + } ); + }, + + "ATTR": function( name, operator, check ) { + return function( elem ) { + var result = Sizzle.attr( elem, name ); + + if ( result == null ) { + return operator === "!="; + } + if ( !operator ) { + return true; + } + + result += ""; + + /* eslint-disable max-len */ + + return operator === "=" ? result === check : + operator === "!=" ? result !== check : + operator === "^=" ? check && result.indexOf( check ) === 0 : + operator === "*=" ? check && result.indexOf( check ) > -1 : + operator === "$=" ? check && result.slice( -check.length ) === check : + operator === "~=" ? ( " " + result.replace( rwhitespace, " " ) + " " ).indexOf( check ) > -1 : + operator === "|=" ? result === check || result.slice( 0, check.length + 1 ) === check + "-" : + false; + /* eslint-enable max-len */ + + }; + }, + + "CHILD": function( type, what, _argument, first, last ) { + var simple = type.slice( 0, 3 ) !== "nth", + forward = type.slice( -4 ) !== "last", + ofType = what === "of-type"; + + return first === 1 && last === 0 ? + + // Shortcut for :nth-*(n) + function( elem ) { + return !!elem.parentNode; + } : + + function( elem, _context, xml ) { + var cache, uniqueCache, outerCache, node, nodeIndex, start, + dir = simple !== forward ? "nextSibling" : "previousSibling", + parent = elem.parentNode, + name = ofType && elem.nodeName.toLowerCase(), + useCache = !xml && !ofType, + diff = false; + + if ( parent ) { + + // :(first|last|only)-(child|of-type) + if ( simple ) { + while ( dir ) { + node = elem; + while ( ( node = node[ dir ] ) ) { + if ( ofType ? + node.nodeName.toLowerCase() === name : + node.nodeType === 1 ) { + + return false; + } + } + + // Reverse direction for :only-* (if we haven't yet done so) + start = dir = type === "only" && !start && "nextSibling"; + } + return true; + } + + start = [ forward ? parent.firstChild : parent.lastChild ]; + + // non-xml :nth-child(...) stores cache data on `parent` + if ( forward && useCache ) { + + // Seek `elem` from a previously-cached index + + // ...in a gzip-friendly way + node = parent; + outerCache = node[ expando ] || ( node[ expando ] = {} ); + + // Support: IE <9 only + // Defend against cloned attroperties (jQuery gh-1709) + uniqueCache = outerCache[ node.uniqueID ] || + ( outerCache[ node.uniqueID ] = {} ); + + cache = uniqueCache[ type ] || []; + nodeIndex = cache[ 0 ] === dirruns && cache[ 1 ]; + diff = nodeIndex && cache[ 2 ]; + node = nodeIndex && parent.childNodes[ nodeIndex ]; + + while ( ( node = ++nodeIndex && node && node[ dir ] || + + // Fallback to seeking `elem` from the start + ( diff = nodeIndex = 0 ) || start.pop() ) ) { + + // When found, cache indexes on `parent` and break + if ( node.nodeType === 1 && ++diff && node === elem ) { + uniqueCache[ type ] = [ dirruns, nodeIndex, diff ]; + break; + } + } + + } else { + + // Use previously-cached element index if available + if ( useCache ) { + + // ...in a gzip-friendly way + node = elem; + outerCache = node[ expando ] || ( node[ expando ] = {} ); + + // Support: IE <9 only + // Defend against cloned attroperties (jQuery gh-1709) + uniqueCache = outerCache[ node.uniqueID ] || + ( outerCache[ node.uniqueID ] = {} ); + + cache = uniqueCache[ type ] || []; + nodeIndex = cache[ 0 ] === dirruns && cache[ 1 ]; + diff = nodeIndex; + } + + // xml :nth-child(...) + // or :nth-last-child(...) or :nth(-last)?-of-type(...) + if ( diff === false ) { + + // Use the same loop as above to seek `elem` from the start + while ( ( node = ++nodeIndex && node && node[ dir ] || + ( diff = nodeIndex = 0 ) || start.pop() ) ) { + + if ( ( ofType ? + node.nodeName.toLowerCase() === name : + node.nodeType === 1 ) && + ++diff ) { + + // Cache the index of each encountered element + if ( useCache ) { + outerCache = node[ expando ] || + ( node[ expando ] = {} ); + + // Support: IE <9 only + // Defend against cloned attroperties (jQuery gh-1709) + uniqueCache = outerCache[ node.uniqueID ] || + ( outerCache[ node.uniqueID ] = {} ); + + uniqueCache[ type ] = [ dirruns, diff ]; + } + + if ( node === elem ) { + break; + } + } + } + } + } + + // Incorporate the offset, then check against cycle size + diff -= last; + return diff === first || ( diff % first === 0 && diff / first >= 0 ); + } + }; + }, + + "PSEUDO": function( pseudo, argument ) { + + // pseudo-class names are case-insensitive + // http://www.w3.org/TR/selectors/#pseudo-classes + // Prioritize by case sensitivity in case custom pseudos are added with uppercase letters + // Remember that setFilters inherits from pseudos + var args, + fn = Expr.pseudos[ pseudo ] || Expr.setFilters[ pseudo.toLowerCase() ] || + Sizzle.error( "unsupported pseudo: " + pseudo ); + + // The user may use createPseudo to indicate that + // arguments are needed to create the filter function + // just as Sizzle does + if ( fn[ expando ] ) { + return fn( argument ); + } + + // But maintain support for old signatures + if ( fn.length > 1 ) { + args = [ pseudo, pseudo, "", argument ]; + return Expr.setFilters.hasOwnProperty( pseudo.toLowerCase() ) ? + markFunction( function( seed, matches ) { + var idx, + matched = fn( seed, argument ), + i = matched.length; + while ( i-- ) { + idx = indexOf( seed, matched[ i ] ); + seed[ idx ] = !( matches[ idx ] = matched[ i ] ); + } + } ) : + function( elem ) { + return fn( elem, 0, args ); + }; + } + + return fn; + } + }, + + pseudos: { + + // Potentially complex pseudos + "not": markFunction( function( selector ) { + + // Trim the selector passed to compile + // to avoid treating leading and trailing + // spaces as combinators + var input = [], + results = [], + matcher = compile( selector.replace( rtrim, "$1" ) ); + + return matcher[ expando ] ? + markFunction( function( seed, matches, _context, xml ) { + var elem, + unmatched = matcher( seed, null, xml, [] ), + i = seed.length; + + // Match elements unmatched by `matcher` + while ( i-- ) { + if ( ( elem = unmatched[ i ] ) ) { + seed[ i ] = !( matches[ i ] = elem ); + } + } + } ) : + function( elem, _context, xml ) { + input[ 0 ] = elem; + matcher( input, null, xml, results ); + + // Don't keep the element (issue #299) + input[ 0 ] = null; + return !results.pop(); + }; + } ), + + "has": markFunction( function( selector ) { + return function( elem ) { + return Sizzle( selector, elem ).length > 0; + }; + } ), + + "contains": markFunction( function( text ) { + text = text.replace( runescape, funescape ); + return function( elem ) { + return ( elem.textContent || getText( elem ) ).indexOf( text ) > -1; + }; + } ), + + // "Whether an element is represented by a :lang() selector + // is based solely on the element's language value + // being equal to the identifier C, + // or beginning with the identifier C immediately followed by "-". + // The matching of C against the element's language value is performed case-insensitively. + // The identifier C does not have to be a valid language name." + // http://www.w3.org/TR/selectors/#lang-pseudo + "lang": markFunction( function( lang ) { + + // lang value must be a valid identifier + if ( !ridentifier.test( lang || "" ) ) { + Sizzle.error( "unsupported lang: " + lang ); + } + lang = lang.replace( runescape, funescape ).toLowerCase(); + return function( elem ) { + var elemLang; + do { + if ( ( elemLang = documentIsHTML ? + elem.lang : + elem.getAttribute( "xml:lang" ) || elem.getAttribute( "lang" ) ) ) { + + elemLang = elemLang.toLowerCase(); + return elemLang === lang || elemLang.indexOf( lang + "-" ) === 0; + } + } while ( ( elem = elem.parentNode ) && elem.nodeType === 1 ); + return false; + }; + } ), + + // Miscellaneous + "target": function( elem ) { + var hash = window.location && window.location.hash; + return hash && hash.slice( 1 ) === elem.id; + }, + + "root": function( elem ) { + return elem === docElem; + }, + + "focus": function( elem ) { + return elem === document.activeElement && + ( !document.hasFocus || document.hasFocus() ) && + !!( elem.type || elem.href || ~elem.tabIndex ); + }, + + // Boolean properties + "enabled": createDisabledPseudo( false ), + "disabled": createDisabledPseudo( true ), + + "checked": function( elem ) { + + // In CSS3, :checked should return both checked and selected elements + // http://www.w3.org/TR/2011/REC-css3-selectors-20110929/#checked + var nodeName = elem.nodeName.toLowerCase(); + return ( nodeName === "input" && !!elem.checked ) || + ( nodeName === "option" && !!elem.selected ); + }, + + "selected": function( elem ) { + + // Accessing this property makes selected-by-default + // options in Safari work properly + if ( elem.parentNode ) { + // eslint-disable-next-line no-unused-expressions + elem.parentNode.selectedIndex; + } + + return elem.selected === true; + }, + + // Contents + "empty": function( elem ) { + + // http://www.w3.org/TR/selectors/#empty-pseudo + // :empty is negated by element (1) or content nodes (text: 3; cdata: 4; entity ref: 5), + // but not by others (comment: 8; processing instruction: 7; etc.) + // nodeType < 6 works because attributes (2) do not appear as children + for ( elem = elem.firstChild; elem; elem = elem.nextSibling ) { + if ( elem.nodeType < 6 ) { + return false; + } + } + return true; + }, + + "parent": function( elem ) { + return !Expr.pseudos[ "empty" ]( elem ); + }, + + // Element/input types + "header": function( elem ) { + return rheader.test( elem.nodeName ); + }, + + "input": function( elem ) { + return rinputs.test( elem.nodeName ); + }, + + "button": function( elem ) { + var name = elem.nodeName.toLowerCase(); + return name === "input" && elem.type === "button" || name === "button"; + }, + + "text": function( elem ) { + var attr; + return elem.nodeName.toLowerCase() === "input" && + elem.type === "text" && + + // Support: IE<8 + // New HTML5 attribute values (e.g., "search") appear with elem.type === "text" + ( ( attr = elem.getAttribute( "type" ) ) == null || + attr.toLowerCase() === "text" ); + }, + + // Position-in-collection + "first": createPositionalPseudo( function() { + return [ 0 ]; + } ), + + "last": createPositionalPseudo( function( _matchIndexes, length ) { + return [ length - 1 ]; + } ), + + "eq": createPositionalPseudo( function( _matchIndexes, length, argument ) { + return [ argument < 0 ? argument + length : argument ]; + } ), + + "even": createPositionalPseudo( function( matchIndexes, length ) { + var i = 0; + for ( ; i < length; i += 2 ) { + matchIndexes.push( i ); + } + return matchIndexes; + } ), + + "odd": createPositionalPseudo( function( matchIndexes, length ) { + var i = 1; + for ( ; i < length; i += 2 ) { + matchIndexes.push( i ); + } + return matchIndexes; + } ), + + "lt": createPositionalPseudo( function( matchIndexes, length, argument ) { + var i = argument < 0 ? + argument + length : + argument > length ? + length : + argument; + for ( ; --i >= 0; ) { + matchIndexes.push( i ); + } + return matchIndexes; + } ), + + "gt": createPositionalPseudo( function( matchIndexes, length, argument ) { + var i = argument < 0 ? argument + length : argument; + for ( ; ++i < length; ) { + matchIndexes.push( i ); + } + return matchIndexes; + } ) + } +}; + +Expr.pseudos[ "nth" ] = Expr.pseudos[ "eq" ]; + +// Add button/input type pseudos +for ( i in { radio: true, checkbox: true, file: true, password: true, image: true } ) { + Expr.pseudos[ i ] = createInputPseudo( i ); +} +for ( i in { submit: true, reset: true } ) { + Expr.pseudos[ i ] = createButtonPseudo( i ); +} + +// Easy API for creating new setFilters +function setFilters() {} +setFilters.prototype = Expr.filters = Expr.pseudos; +Expr.setFilters = new setFilters(); + +tokenize = Sizzle.tokenize = function( selector, parseOnly ) { + var matched, match, tokens, type, + soFar, groups, preFilters, + cached = tokenCache[ selector + " " ]; + + if ( cached ) { + return parseOnly ? 0 : cached.slice( 0 ); + } + + soFar = selector; + groups = []; + preFilters = Expr.preFilter; + + while ( soFar ) { + + // Comma and first run + if ( !matched || ( match = rcomma.exec( soFar ) ) ) { + if ( match ) { + + // Don't consume trailing commas as valid + soFar = soFar.slice( match[ 0 ].length ) || soFar; + } + groups.push( ( tokens = [] ) ); + } + + matched = false; + + // Combinators + if ( ( match = rcombinators.exec( soFar ) ) ) { + matched = match.shift(); + tokens.push( { + value: matched, + + // Cast descendant combinators to space + type: match[ 0 ].replace( rtrim, " " ) + } ); + soFar = soFar.slice( matched.length ); + } + + // Filters + for ( type in Expr.filter ) { + if ( ( match = matchExpr[ type ].exec( soFar ) ) && ( !preFilters[ type ] || + ( match = preFilters[ type ]( match ) ) ) ) { + matched = match.shift(); + tokens.push( { + value: matched, + type: type, + matches: match + } ); + soFar = soFar.slice( matched.length ); + } + } + + if ( !matched ) { + break; + } + } + + // Return the length of the invalid excess + // if we're just parsing + // Otherwise, throw an error or return tokens + return parseOnly ? + soFar.length : + soFar ? + Sizzle.error( selector ) : + + // Cache the tokens + tokenCache( selector, groups ).slice( 0 ); +}; + +function toSelector( tokens ) { + var i = 0, + len = tokens.length, + selector = ""; + for ( ; i < len; i++ ) { + selector += tokens[ i ].value; + } + return selector; +} + +function addCombinator( matcher, combinator, base ) { + var dir = combinator.dir, + skip = combinator.next, + key = skip || dir, + checkNonElements = base && key === "parentNode", + doneName = done++; + + return combinator.first ? + + // Check against closest ancestor/preceding element + function( elem, context, xml ) { + while ( ( elem = elem[ dir ] ) ) { + if ( elem.nodeType === 1 || checkNonElements ) { + return matcher( elem, context, xml ); + } + } + return false; + } : + + // Check against all ancestor/preceding elements + function( elem, context, xml ) { + var oldCache, uniqueCache, outerCache, + newCache = [ dirruns, doneName ]; + + // We can't set arbitrary data on XML nodes, so they don't benefit from combinator caching + if ( xml ) { + while ( ( elem = elem[ dir ] ) ) { + if ( elem.nodeType === 1 || checkNonElements ) { + if ( matcher( elem, context, xml ) ) { + return true; + } + } + } + } else { + while ( ( elem = elem[ dir ] ) ) { + if ( elem.nodeType === 1 || checkNonElements ) { + outerCache = elem[ expando ] || ( elem[ expando ] = {} ); + + // Support: IE <9 only + // Defend against cloned attroperties (jQuery gh-1709) + uniqueCache = outerCache[ elem.uniqueID ] || + ( outerCache[ elem.uniqueID ] = {} ); + + if ( skip && skip === elem.nodeName.toLowerCase() ) { + elem = elem[ dir ] || elem; + } else if ( ( oldCache = uniqueCache[ key ] ) && + oldCache[ 0 ] === dirruns && oldCache[ 1 ] === doneName ) { + + // Assign to newCache so results back-propagate to previous elements + return ( newCache[ 2 ] = oldCache[ 2 ] ); + } else { + + // Reuse newcache so results back-propagate to previous elements + uniqueCache[ key ] = newCache; + + // A match means we're done; a fail means we have to keep checking + if ( ( newCache[ 2 ] = matcher( elem, context, xml ) ) ) { + return true; + } + } + } + } + } + return false; + }; +} + +function elementMatcher( matchers ) { + return matchers.length > 1 ? + function( elem, context, xml ) { + var i = matchers.length; + while ( i-- ) { + if ( !matchers[ i ]( elem, context, xml ) ) { + return false; + } + } + return true; + } : + matchers[ 0 ]; +} + +function multipleContexts( selector, contexts, results ) { + var i = 0, + len = contexts.length; + for ( ; i < len; i++ ) { + Sizzle( selector, contexts[ i ], results ); + } + return results; +} + +function condense( unmatched, map, filter, context, xml ) { + var elem, + newUnmatched = [], + i = 0, + len = unmatched.length, + mapped = map != null; + + for ( ; i < len; i++ ) { + if ( ( elem = unmatched[ i ] ) ) { + if ( !filter || filter( elem, context, xml ) ) { + newUnmatched.push( elem ); + if ( mapped ) { + map.push( i ); + } + } + } + } + + return newUnmatched; +} + +function setMatcher( preFilter, selector, matcher, postFilter, postFinder, postSelector ) { + if ( postFilter && !postFilter[ expando ] ) { + postFilter = setMatcher( postFilter ); + } + if ( postFinder && !postFinder[ expando ] ) { + postFinder = setMatcher( postFinder, postSelector ); + } + return markFunction( function( seed, results, context, xml ) { + var temp, i, elem, + preMap = [], + postMap = [], + preexisting = results.length, + + // Get initial elements from seed or context + elems = seed || multipleContexts( + selector || "*", + context.nodeType ? [ context ] : context, + [] + ), + + // Prefilter to get matcher input, preserving a map for seed-results synchronization + matcherIn = preFilter && ( seed || !selector ) ? + condense( elems, preMap, preFilter, context, xml ) : + elems, + + matcherOut = matcher ? + + // If we have a postFinder, or filtered seed, or non-seed postFilter or preexisting results, + postFinder || ( seed ? preFilter : preexisting || postFilter ) ? + + // ...intermediate processing is necessary + [] : + + // ...otherwise use results directly + results : + matcherIn; + + // Find primary matches + if ( matcher ) { + matcher( matcherIn, matcherOut, context, xml ); + } + + // Apply postFilter + if ( postFilter ) { + temp = condense( matcherOut, postMap ); + postFilter( temp, [], context, xml ); + + // Un-match failing elements by moving them back to matcherIn + i = temp.length; + while ( i-- ) { + if ( ( elem = temp[ i ] ) ) { + matcherOut[ postMap[ i ] ] = !( matcherIn[ postMap[ i ] ] = elem ); + } + } + } + + if ( seed ) { + if ( postFinder || preFilter ) { + if ( postFinder ) { + + // Get the final matcherOut by condensing this intermediate into postFinder contexts + temp = []; + i = matcherOut.length; + while ( i-- ) { + if ( ( elem = matcherOut[ i ] ) ) { + + // Restore matcherIn since elem is not yet a final match + temp.push( ( matcherIn[ i ] = elem ) ); + } + } + postFinder( null, ( matcherOut = [] ), temp, xml ); + } + + // Move matched elements from seed to results to keep them synchronized + i = matcherOut.length; + while ( i-- ) { + if ( ( elem = matcherOut[ i ] ) && + ( temp = postFinder ? indexOf( seed, elem ) : preMap[ i ] ) > -1 ) { + + seed[ temp ] = !( results[ temp ] = elem ); + } + } + } + + // Add elements to results, through postFinder if defined + } else { + matcherOut = condense( + matcherOut === results ? + matcherOut.splice( preexisting, matcherOut.length ) : + matcherOut + ); + if ( postFinder ) { + postFinder( null, results, matcherOut, xml ); + } else { + push.apply( results, matcherOut ); + } + } + } ); +} + +function matcherFromTokens( tokens ) { + var checkContext, matcher, j, + len = tokens.length, + leadingRelative = Expr.relative[ tokens[ 0 ].type ], + implicitRelative = leadingRelative || Expr.relative[ " " ], + i = leadingRelative ? 1 : 0, + + // The foundational matcher ensures that elements are reachable from top-level context(s) + matchContext = addCombinator( function( elem ) { + return elem === checkContext; + }, implicitRelative, true ), + matchAnyContext = addCombinator( function( elem ) { + return indexOf( checkContext, elem ) > -1; + }, implicitRelative, true ), + matchers = [ function( elem, context, xml ) { + var ret = ( !leadingRelative && ( xml || context !== outermostContext ) ) || ( + ( checkContext = context ).nodeType ? + matchContext( elem, context, xml ) : + matchAnyContext( elem, context, xml ) ); + + // Avoid hanging onto element (issue #299) + checkContext = null; + return ret; + } ]; + + for ( ; i < len; i++ ) { + if ( ( matcher = Expr.relative[ tokens[ i ].type ] ) ) { + matchers = [ addCombinator( elementMatcher( matchers ), matcher ) ]; + } else { + matcher = Expr.filter[ tokens[ i ].type ].apply( null, tokens[ i ].matches ); + + // Return special upon seeing a positional matcher + if ( matcher[ expando ] ) { + + // Find the next relative operator (if any) for proper handling + j = ++i; + for ( ; j < len; j++ ) { + if ( Expr.relative[ tokens[ j ].type ] ) { + break; + } + } + return setMatcher( + i > 1 && elementMatcher( matchers ), + i > 1 && toSelector( + + // If the preceding token was a descendant combinator, insert an implicit any-element `*` + tokens + .slice( 0, i - 1 ) + .concat( { value: tokens[ i - 2 ].type === " " ? "*" : "" } ) + ).replace( rtrim, "$1" ), + matcher, + i < j && matcherFromTokens( tokens.slice( i, j ) ), + j < len && matcherFromTokens( ( tokens = tokens.slice( j ) ) ), + j < len && toSelector( tokens ) + ); + } + matchers.push( matcher ); + } + } + + return elementMatcher( matchers ); +} + +function matcherFromGroupMatchers( elementMatchers, setMatchers ) { + var bySet = setMatchers.length > 0, + byElement = elementMatchers.length > 0, + superMatcher = function( seed, context, xml, results, outermost ) { + var elem, j, matcher, + matchedCount = 0, + i = "0", + unmatched = seed && [], + setMatched = [], + contextBackup = outermostContext, + + // We must always have either seed elements or outermost context + elems = seed || byElement && Expr.find[ "TAG" ]( "*", outermost ), + + // Use integer dirruns iff this is the outermost matcher + dirrunsUnique = ( dirruns += contextBackup == null ? 1 : Math.random() || 0.1 ), + len = elems.length; + + if ( outermost ) { + + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + // eslint-disable-next-line eqeqeq + outermostContext = context == document || context || outermost; + } + + // Add elements passing elementMatchers directly to results + // Support: IE<9, Safari + // Tolerate NodeList properties (IE: "length"; Safari: ) matching elements by id + for ( ; i !== len && ( elem = elems[ i ] ) != null; i++ ) { + if ( byElement && elem ) { + j = 0; + + // Support: IE 11+, Edge 17 - 18+ + // IE/Edge sometimes throw a "Permission denied" error when strict-comparing + // two documents; shallow comparisons work. + // eslint-disable-next-line eqeqeq + if ( !context && elem.ownerDocument != document ) { + setDocument( elem ); + xml = !documentIsHTML; + } + while ( ( matcher = elementMatchers[ j++ ] ) ) { + if ( matcher( elem, context || document, xml ) ) { + results.push( elem ); + break; + } + } + if ( outermost ) { + dirruns = dirrunsUnique; + } + } + + // Track unmatched elements for set filters + if ( bySet ) { + + // They will have gone through all possible matchers + if ( ( elem = !matcher && elem ) ) { + matchedCount--; + } + + // Lengthen the array for every element, matched or not + if ( seed ) { + unmatched.push( elem ); + } + } + } + + // `i` is now the count of elements visited above, and adding it to `matchedCount` + // makes the latter nonnegative. + matchedCount += i; + + // Apply set filters to unmatched elements + // NOTE: This can be skipped if there are no unmatched elements (i.e., `matchedCount` + // equals `i`), unless we didn't visit _any_ elements in the above loop because we have + // no element matchers and no seed. + // Incrementing an initially-string "0" `i` allows `i` to remain a string only in that + // case, which will result in a "00" `matchedCount` that differs from `i` but is also + // numerically zero. + if ( bySet && i !== matchedCount ) { + j = 0; + while ( ( matcher = setMatchers[ j++ ] ) ) { + matcher( unmatched, setMatched, context, xml ); + } + + if ( seed ) { + + // Reintegrate element matches to eliminate the need for sorting + if ( matchedCount > 0 ) { + while ( i-- ) { + if ( !( unmatched[ i ] || setMatched[ i ] ) ) { + setMatched[ i ] = pop.call( results ); + } + } + } + + // Discard index placeholder values to get only actual matches + setMatched = condense( setMatched ); + } + + // Add matches to results + push.apply( results, setMatched ); + + // Seedless set matches succeeding multiple successful matchers stipulate sorting + if ( outermost && !seed && setMatched.length > 0 && + ( matchedCount + setMatchers.length ) > 1 ) { + + Sizzle.uniqueSort( results ); + } + } + + // Override manipulation of globals by nested matchers + if ( outermost ) { + dirruns = dirrunsUnique; + outermostContext = contextBackup; + } + + return unmatched; + }; + + return bySet ? + markFunction( superMatcher ) : + superMatcher; +} + +compile = Sizzle.compile = function( selector, match /* Internal Use Only */ ) { + var i, + setMatchers = [], + elementMatchers = [], + cached = compilerCache[ selector + " " ]; + + if ( !cached ) { + + // Generate a function of recursive functions that can be used to check each element + if ( !match ) { + match = tokenize( selector ); + } + i = match.length; + while ( i-- ) { + cached = matcherFromTokens( match[ i ] ); + if ( cached[ expando ] ) { + setMatchers.push( cached ); + } else { + elementMatchers.push( cached ); + } + } + + // Cache the compiled function + cached = compilerCache( + selector, + matcherFromGroupMatchers( elementMatchers, setMatchers ) + ); + + // Save selector and tokenization + cached.selector = selector; + } + return cached; +}; + +/** + * A low-level selection function that works with Sizzle's compiled + * selector functions + * @param {String|Function} selector A selector or a pre-compiled + * selector function built with Sizzle.compile + * @param {Element} context + * @param {Array} [results] + * @param {Array} [seed] A set of elements to match against + */ +select = Sizzle.select = function( selector, context, results, seed ) { + var i, tokens, token, type, find, + compiled = typeof selector === "function" && selector, + match = !seed && tokenize( ( selector = compiled.selector || selector ) ); + + results = results || []; + + // Try to minimize operations if there is only one selector in the list and no seed + // (the latter of which guarantees us context) + if ( match.length === 1 ) { + + // Reduce context if the leading compound selector is an ID + tokens = match[ 0 ] = match[ 0 ].slice( 0 ); + if ( tokens.length > 2 && ( token = tokens[ 0 ] ).type === "ID" && + context.nodeType === 9 && documentIsHTML && Expr.relative[ tokens[ 1 ].type ] ) { + + context = ( Expr.find[ "ID" ]( token.matches[ 0 ] + .replace( runescape, funescape ), context ) || [] )[ 0 ]; + if ( !context ) { + return results; + + // Precompiled matchers will still verify ancestry, so step up a level + } else if ( compiled ) { + context = context.parentNode; + } + + selector = selector.slice( tokens.shift().value.length ); + } + + // Fetch a seed set for right-to-left matching + i = matchExpr[ "needsContext" ].test( selector ) ? 0 : tokens.length; + while ( i-- ) { + token = tokens[ i ]; + + // Abort if we hit a combinator + if ( Expr.relative[ ( type = token.type ) ] ) { + break; + } + if ( ( find = Expr.find[ type ] ) ) { + + // Search, expanding context for leading sibling combinators + if ( ( seed = find( + token.matches[ 0 ].replace( runescape, funescape ), + rsibling.test( tokens[ 0 ].type ) && testContext( context.parentNode ) || + context + ) ) ) { + + // If seed is empty or no tokens remain, we can return early + tokens.splice( i, 1 ); + selector = seed.length && toSelector( tokens ); + if ( !selector ) { + push.apply( results, seed ); + return results; + } + + break; + } + } + } + } + + // Compile and execute a filtering function if one is not provided + // Provide `match` to avoid retokenization if we modified the selector above + ( compiled || compile( selector, match ) )( + seed, + context, + !documentIsHTML, + results, + !context || rsibling.test( selector ) && testContext( context.parentNode ) || context + ); + return results; +}; + +// One-time assignments + +// Sort stability +support.sortStable = expando.split( "" ).sort( sortOrder ).join( "" ) === expando; + +// Support: Chrome 14-35+ +// Always assume duplicates if they aren't passed to the comparison function +support.detectDuplicates = !!hasDuplicate; + +// Initialize against the default document +setDocument(); + +// Support: Webkit<537.32 - Safari 6.0.3/Chrome 25 (fixed in Chrome 27) +// Detached nodes confoundingly follow *each other* +support.sortDetached = assert( function( el ) { + + // Should return 1, but returns 4 (following) + return el.compareDocumentPosition( document.createElement( "fieldset" ) ) & 1; +} ); + +// Support: IE<8 +// Prevent attribute/property "interpolation" +// https://msdn.microsoft.com/en-us/library/ms536429%28VS.85%29.aspx +if ( !assert( function( el ) { + el.innerHTML = ""; + return el.firstChild.getAttribute( "href" ) === "#"; +} ) ) { + addHandle( "type|href|height|width", function( elem, name, isXML ) { + if ( !isXML ) { + return elem.getAttribute( name, name.toLowerCase() === "type" ? 1 : 2 ); + } + } ); +} + +// Support: IE<9 +// Use defaultValue in place of getAttribute("value") +if ( !support.attributes || !assert( function( el ) { + el.innerHTML = ""; + el.firstChild.setAttribute( "value", "" ); + return el.firstChild.getAttribute( "value" ) === ""; +} ) ) { + addHandle( "value", function( elem, _name, isXML ) { + if ( !isXML && elem.nodeName.toLowerCase() === "input" ) { + return elem.defaultValue; + } + } ); +} + +// Support: IE<9 +// Use getAttributeNode to fetch booleans when getAttribute lies +if ( !assert( function( el ) { + return el.getAttribute( "disabled" ) == null; +} ) ) { + addHandle( booleans, function( elem, name, isXML ) { + var val; + if ( !isXML ) { + return elem[ name ] === true ? name.toLowerCase() : + ( val = elem.getAttributeNode( name ) ) && val.specified ? + val.value : + null; + } + } ); +} + +return Sizzle; + +} )( window ); + + + +jQuery.find = Sizzle; +jQuery.expr = Sizzle.selectors; + +// Deprecated +jQuery.expr[ ":" ] = jQuery.expr.pseudos; +jQuery.uniqueSort = jQuery.unique = Sizzle.uniqueSort; +jQuery.text = Sizzle.getText; +jQuery.isXMLDoc = Sizzle.isXML; +jQuery.contains = Sizzle.contains; +jQuery.escapeSelector = Sizzle.escape; + + + + +var dir = function( elem, dir, until ) { + var matched = [], + truncate = until !== undefined; + + while ( ( elem = elem[ dir ] ) && elem.nodeType !== 9 ) { + if ( elem.nodeType === 1 ) { + if ( truncate && jQuery( elem ).is( until ) ) { + break; + } + matched.push( elem ); + } + } + return matched; +}; + + +var siblings = function( n, elem ) { + var matched = []; + + for ( ; n; n = n.nextSibling ) { + if ( n.nodeType === 1 && n !== elem ) { + matched.push( n ); + } + } + + return matched; +}; + + +var rneedsContext = jQuery.expr.match.needsContext; + + + +function nodeName( elem, name ) { + + return elem.nodeName && elem.nodeName.toLowerCase() === name.toLowerCase(); + +}; +var rsingleTag = ( /^<([a-z][^\/\0>:\x20\t\r\n\f]*)[\x20\t\r\n\f]*\/?>(?:<\/\1>|)$/i ); + + + +// Implement the identical functionality for filter and not +function winnow( elements, qualifier, not ) { + if ( isFunction( qualifier ) ) { + return jQuery.grep( elements, function( elem, i ) { + return !!qualifier.call( elem, i, elem ) !== not; + } ); + } + + // Single element + if ( qualifier.nodeType ) { + return jQuery.grep( elements, function( elem ) { + return ( elem === qualifier ) !== not; + } ); + } + + // Arraylike of elements (jQuery, arguments, Array) + if ( typeof qualifier !== "string" ) { + return jQuery.grep( elements, function( elem ) { + return ( indexOf.call( qualifier, elem ) > -1 ) !== not; + } ); + } + + // Filtered directly for both simple and complex selectors + return jQuery.filter( qualifier, elements, not ); +} + +jQuery.filter = function( expr, elems, not ) { + var elem = elems[ 0 ]; + + if ( not ) { + expr = ":not(" + expr + ")"; + } + + if ( elems.length === 1 && elem.nodeType === 1 ) { + return jQuery.find.matchesSelector( elem, expr ) ? [ elem ] : []; + } + + return jQuery.find.matches( expr, jQuery.grep( elems, function( elem ) { + return elem.nodeType === 1; + } ) ); +}; + +jQuery.fn.extend( { + find: function( selector ) { + var i, ret, + len = this.length, + self = this; + + if ( typeof selector !== "string" ) { + return this.pushStack( jQuery( selector ).filter( function() { + for ( i = 0; i < len; i++ ) { + if ( jQuery.contains( self[ i ], this ) ) { + return true; + } + } + } ) ); + } + + ret = this.pushStack( [] ); + + for ( i = 0; i < len; i++ ) { + jQuery.find( selector, self[ i ], ret ); + } + + return len > 1 ? jQuery.uniqueSort( ret ) : ret; + }, + filter: function( selector ) { + return this.pushStack( winnow( this, selector || [], false ) ); + }, + not: function( selector ) { + return this.pushStack( winnow( this, selector || [], true ) ); + }, + is: function( selector ) { + return !!winnow( + this, + + // If this is a positional/relative selector, check membership in the returned set + // so $("p:first").is("p:last") won't return true for a doc with two "p". + typeof selector === "string" && rneedsContext.test( selector ) ? + jQuery( selector ) : + selector || [], + false + ).length; + } +} ); + + +// Initialize a jQuery object + + +// A central reference to the root jQuery(document) +var rootjQuery, + + // A simple way to check for HTML strings + // Prioritize #id over to avoid XSS via location.hash (#9521) + // Strict HTML recognition (#11290: must start with <) + // Shortcut simple #id case for speed + rquickExpr = /^(?:\s*(<[\w\W]+>)[^>]*|#([\w-]+))$/, + + init = jQuery.fn.init = function( selector, context, root ) { + var match, elem; + + // HANDLE: $(""), $(null), $(undefined), $(false) + if ( !selector ) { + return this; + } + + // Method init() accepts an alternate rootjQuery + // so migrate can support jQuery.sub (gh-2101) + root = root || rootjQuery; + + // Handle HTML strings + if ( typeof selector === "string" ) { + if ( selector[ 0 ] === "<" && + selector[ selector.length - 1 ] === ">" && + selector.length >= 3 ) { + + // Assume that strings that start and end with <> are HTML and skip the regex check + match = [ null, selector, null ]; + + } else { + match = rquickExpr.exec( selector ); + } + + // Match html or make sure no context is specified for #id + if ( match && ( match[ 1 ] || !context ) ) { + + // HANDLE: $(html) -> $(array) + if ( match[ 1 ] ) { + context = context instanceof jQuery ? context[ 0 ] : context; + + // Option to run scripts is true for back-compat + // Intentionally let the error be thrown if parseHTML is not present + jQuery.merge( this, jQuery.parseHTML( + match[ 1 ], + context && context.nodeType ? context.ownerDocument || context : document, + true + ) ); + + // HANDLE: $(html, props) + if ( rsingleTag.test( match[ 1 ] ) && jQuery.isPlainObject( context ) ) { + for ( match in context ) { + + // Properties of context are called as methods if possible + if ( isFunction( this[ match ] ) ) { + this[ match ]( context[ match ] ); + + // ...and otherwise set as attributes + } else { + this.attr( match, context[ match ] ); + } + } + } + + return this; + + // HANDLE: $(#id) + } else { + elem = document.getElementById( match[ 2 ] ); + + if ( elem ) { + + // Inject the element directly into the jQuery object + this[ 0 ] = elem; + this.length = 1; + } + return this; + } + + // HANDLE: $(expr, $(...)) + } else if ( !context || context.jquery ) { + return ( context || root ).find( selector ); + + // HANDLE: $(expr, context) + // (which is just equivalent to: $(context).find(expr) + } else { + return this.constructor( context ).find( selector ); + } + + // HANDLE: $(DOMElement) + } else if ( selector.nodeType ) { + this[ 0 ] = selector; + this.length = 1; + return this; + + // HANDLE: $(function) + // Shortcut for document ready + } else if ( isFunction( selector ) ) { + return root.ready !== undefined ? + root.ready( selector ) : + + // Execute immediately if ready is not present + selector( jQuery ); + } + + return jQuery.makeArray( selector, this ); + }; + +// Give the init function the jQuery prototype for later instantiation +init.prototype = jQuery.fn; + +// Initialize central reference +rootjQuery = jQuery( document ); + + +var rparentsprev = /^(?:parents|prev(?:Until|All))/, + + // Methods guaranteed to produce a unique set when starting from a unique set + guaranteedUnique = { + children: true, + contents: true, + next: true, + prev: true + }; + +jQuery.fn.extend( { + has: function( target ) { + var targets = jQuery( target, this ), + l = targets.length; + + return this.filter( function() { + var i = 0; + for ( ; i < l; i++ ) { + if ( jQuery.contains( this, targets[ i ] ) ) { + return true; + } + } + } ); + }, + + closest: function( selectors, context ) { + var cur, + i = 0, + l = this.length, + matched = [], + targets = typeof selectors !== "string" && jQuery( selectors ); + + // Positional selectors never match, since there's no _selection_ context + if ( !rneedsContext.test( selectors ) ) { + for ( ; i < l; i++ ) { + for ( cur = this[ i ]; cur && cur !== context; cur = cur.parentNode ) { + + // Always skip document fragments + if ( cur.nodeType < 11 && ( targets ? + targets.index( cur ) > -1 : + + // Don't pass non-elements to Sizzle + cur.nodeType === 1 && + jQuery.find.matchesSelector( cur, selectors ) ) ) { + + matched.push( cur ); + break; + } + } + } + } + + return this.pushStack( matched.length > 1 ? jQuery.uniqueSort( matched ) : matched ); + }, + + // Determine the position of an element within the set + index: function( elem ) { + + // No argument, return index in parent + if ( !elem ) { + return ( this[ 0 ] && this[ 0 ].parentNode ) ? this.first().prevAll().length : -1; + } + + // Index in selector + if ( typeof elem === "string" ) { + return indexOf.call( jQuery( elem ), this[ 0 ] ); + } + + // Locate the position of the desired element + return indexOf.call( this, + + // If it receives a jQuery object, the first element is used + elem.jquery ? elem[ 0 ] : elem + ); + }, + + add: function( selector, context ) { + return this.pushStack( + jQuery.uniqueSort( + jQuery.merge( this.get(), jQuery( selector, context ) ) + ) + ); + }, + + addBack: function( selector ) { + return this.add( selector == null ? + this.prevObject : this.prevObject.filter( selector ) + ); + } +} ); + +function sibling( cur, dir ) { + while ( ( cur = cur[ dir ] ) && cur.nodeType !== 1 ) {} + return cur; +} + +jQuery.each( { + parent: function( elem ) { + var parent = elem.parentNode; + return parent && parent.nodeType !== 11 ? parent : null; + }, + parents: function( elem ) { + return dir( elem, "parentNode" ); + }, + parentsUntil: function( elem, _i, until ) { + return dir( elem, "parentNode", until ); + }, + next: function( elem ) { + return sibling( elem, "nextSibling" ); + }, + prev: function( elem ) { + return sibling( elem, "previousSibling" ); + }, + nextAll: function( elem ) { + return dir( elem, "nextSibling" ); + }, + prevAll: function( elem ) { + return dir( elem, "previousSibling" ); + }, + nextUntil: function( elem, _i, until ) { + return dir( elem, "nextSibling", until ); + }, + prevUntil: function( elem, _i, until ) { + return dir( elem, "previousSibling", until ); + }, + siblings: function( elem ) { + return siblings( ( elem.parentNode || {} ).firstChild, elem ); + }, + children: function( elem ) { + return siblings( elem.firstChild ); + }, + contents: function( elem ) { + if ( elem.contentDocument != null && + + // Support: IE 11+ + // elements with no `data` attribute has an object + // `contentDocument` with a `null` prototype. + getProto( elem.contentDocument ) ) { + + return elem.contentDocument; + } + + // Support: IE 9 - 11 only, iOS 7 only, Android Browser <=4.3 only + // Treat the template element as a regular one in browsers that + // don't support it. + if ( nodeName( elem, "template" ) ) { + elem = elem.content || elem; + } + + return jQuery.merge( [], elem.childNodes ); + } +}, function( name, fn ) { + jQuery.fn[ name ] = function( until, selector ) { + var matched = jQuery.map( this, fn, until ); + + if ( name.slice( -5 ) !== "Until" ) { + selector = until; + } + + if ( selector && typeof selector === "string" ) { + matched = jQuery.filter( selector, matched ); + } + + if ( this.length > 1 ) { + + // Remove duplicates + if ( !guaranteedUnique[ name ] ) { + jQuery.uniqueSort( matched ); + } + + // Reverse order for parents* and prev-derivatives + if ( rparentsprev.test( name ) ) { + matched.reverse(); + } + } + + return this.pushStack( matched ); + }; +} ); +var rnothtmlwhite = ( /[^\x20\t\r\n\f]+/g ); + + + +// Convert String-formatted options into Object-formatted ones +function createOptions( options ) { + var object = {}; + jQuery.each( options.match( rnothtmlwhite ) || [], function( _, flag ) { + object[ flag ] = true; + } ); + return object; +} + +/* + * Create a callback list using the following parameters: + * + * options: an optional list of space-separated options that will change how + * the callback list behaves or a more traditional option object + * + * By default a callback list will act like an event callback list and can be + * "fired" multiple times. + * + * Possible options: + * + * once: will ensure the callback list can only be fired once (like a Deferred) + * + * memory: will keep track of previous values and will call any callback added + * after the list has been fired right away with the latest "memorized" + * values (like a Deferred) + * + * unique: will ensure a callback can only be added once (no duplicate in the list) + * + * stopOnFalse: interrupt callings when a callback returns false + * + */ +jQuery.Callbacks = function( options ) { + + // Convert options from String-formatted to Object-formatted if needed + // (we check in cache first) + options = typeof options === "string" ? + createOptions( options ) : + jQuery.extend( {}, options ); + + var // Flag to know if list is currently firing + firing, + + // Last fire value for non-forgettable lists + memory, + + // Flag to know if list was already fired + fired, + + // Flag to prevent firing + locked, + + // Actual callback list + list = [], + + // Queue of execution data for repeatable lists + queue = [], + + // Index of currently firing callback (modified by add/remove as needed) + firingIndex = -1, + + // Fire callbacks + fire = function() { + + // Enforce single-firing + locked = locked || options.once; + + // Execute callbacks for all pending executions, + // respecting firingIndex overrides and runtime changes + fired = firing = true; + for ( ; queue.length; firingIndex = -1 ) { + memory = queue.shift(); + while ( ++firingIndex < list.length ) { + + // Run callback and check for early termination + if ( list[ firingIndex ].apply( memory[ 0 ], memory[ 1 ] ) === false && + options.stopOnFalse ) { + + // Jump to end and forget the data so .add doesn't re-fire + firingIndex = list.length; + memory = false; + } + } + } + + // Forget the data if we're done with it + if ( !options.memory ) { + memory = false; + } + + firing = false; + + // Clean up if we're done firing for good + if ( locked ) { + + // Keep an empty list if we have data for future add calls + if ( memory ) { + list = []; + + // Otherwise, this object is spent + } else { + list = ""; + } + } + }, + + // Actual Callbacks object + self = { + + // Add a callback or a collection of callbacks to the list + add: function() { + if ( list ) { + + // If we have memory from a past run, we should fire after adding + if ( memory && !firing ) { + firingIndex = list.length - 1; + queue.push( memory ); + } + + ( function add( args ) { + jQuery.each( args, function( _, arg ) { + if ( isFunction( arg ) ) { + if ( !options.unique || !self.has( arg ) ) { + list.push( arg ); + } + } else if ( arg && arg.length && toType( arg ) !== "string" ) { + + // Inspect recursively + add( arg ); + } + } ); + } )( arguments ); + + if ( memory && !firing ) { + fire(); + } + } + return this; + }, + + // Remove a callback from the list + remove: function() { + jQuery.each( arguments, function( _, arg ) { + var index; + while ( ( index = jQuery.inArray( arg, list, index ) ) > -1 ) { + list.splice( index, 1 ); + + // Handle firing indexes + if ( index <= firingIndex ) { + firingIndex--; + } + } + } ); + return this; + }, + + // Check if a given callback is in the list. + // If no argument is given, return whether or not list has callbacks attached. + has: function( fn ) { + return fn ? + jQuery.inArray( fn, list ) > -1 : + list.length > 0; + }, + + // Remove all callbacks from the list + empty: function() { + if ( list ) { + list = []; + } + return this; + }, + + // Disable .fire and .add + // Abort any current/pending executions + // Clear all callbacks and values + disable: function() { + locked = queue = []; + list = memory = ""; + return this; + }, + disabled: function() { + return !list; + }, + + // Disable .fire + // Also disable .add unless we have memory (since it would have no effect) + // Abort any pending executions + lock: function() { + locked = queue = []; + if ( !memory && !firing ) { + list = memory = ""; + } + return this; + }, + locked: function() { + return !!locked; + }, + + // Call all callbacks with the given context and arguments + fireWith: function( context, args ) { + if ( !locked ) { + args = args || []; + args = [ context, args.slice ? args.slice() : args ]; + queue.push( args ); + if ( !firing ) { + fire(); + } + } + return this; + }, + + // Call all the callbacks with the given arguments + fire: function() { + self.fireWith( this, arguments ); + return this; + }, + + // To know if the callbacks have already been called at least once + fired: function() { + return !!fired; + } + }; + + return self; +}; + + +function Identity( v ) { + return v; +} +function Thrower( ex ) { + throw ex; +} + +function adoptValue( value, resolve, reject, noValue ) { + var method; + + try { + + // Check for promise aspect first to privilege synchronous behavior + if ( value && isFunction( ( method = value.promise ) ) ) { + method.call( value ).done( resolve ).fail( reject ); + + // Other thenables + } else if ( value && isFunction( ( method = value.then ) ) ) { + method.call( value, resolve, reject ); + + // Other non-thenables + } else { + + // Control `resolve` arguments by letting Array#slice cast boolean `noValue` to integer: + // * false: [ value ].slice( 0 ) => resolve( value ) + // * true: [ value ].slice( 1 ) => resolve() + resolve.apply( undefined, [ value ].slice( noValue ) ); + } + + // For Promises/A+, convert exceptions into rejections + // Since jQuery.when doesn't unwrap thenables, we can skip the extra checks appearing in + // Deferred#then to conditionally suppress rejection. + } catch ( value ) { + + // Support: Android 4.0 only + // Strict mode functions invoked without .call/.apply get global-object context + reject.apply( undefined, [ value ] ); + } +} + +jQuery.extend( { + + Deferred: function( func ) { + var tuples = [ + + // action, add listener, callbacks, + // ... .then handlers, argument index, [final state] + [ "notify", "progress", jQuery.Callbacks( "memory" ), + jQuery.Callbacks( "memory" ), 2 ], + [ "resolve", "done", jQuery.Callbacks( "once memory" ), + jQuery.Callbacks( "once memory" ), 0, "resolved" ], + [ "reject", "fail", jQuery.Callbacks( "once memory" ), + jQuery.Callbacks( "once memory" ), 1, "rejected" ] + ], + state = "pending", + promise = { + state: function() { + return state; + }, + always: function() { + deferred.done( arguments ).fail( arguments ); + return this; + }, + "catch": function( fn ) { + return promise.then( null, fn ); + }, + + // Keep pipe for back-compat + pipe: function( /* fnDone, fnFail, fnProgress */ ) { + var fns = arguments; + + return jQuery.Deferred( function( newDefer ) { + jQuery.each( tuples, function( _i, tuple ) { + + // Map tuples (progress, done, fail) to arguments (done, fail, progress) + var fn = isFunction( fns[ tuple[ 4 ] ] ) && fns[ tuple[ 4 ] ]; + + // deferred.progress(function() { bind to newDefer or newDefer.notify }) + // deferred.done(function() { bind to newDefer or newDefer.resolve }) + // deferred.fail(function() { bind to newDefer or newDefer.reject }) + deferred[ tuple[ 1 ] ]( function() { + var returned = fn && fn.apply( this, arguments ); + if ( returned && isFunction( returned.promise ) ) { + returned.promise() + .progress( newDefer.notify ) + .done( newDefer.resolve ) + .fail( newDefer.reject ); + } else { + newDefer[ tuple[ 0 ] + "With" ]( + this, + fn ? [ returned ] : arguments + ); + } + } ); + } ); + fns = null; + } ).promise(); + }, + then: function( onFulfilled, onRejected, onProgress ) { + var maxDepth = 0; + function resolve( depth, deferred, handler, special ) { + return function() { + var that = this, + args = arguments, + mightThrow = function() { + var returned, then; + + // Support: Promises/A+ section 2.3.3.3.3 + // https://promisesaplus.com/#point-59 + // Ignore double-resolution attempts + if ( depth < maxDepth ) { + return; + } + + returned = handler.apply( that, args ); + + // Support: Promises/A+ section 2.3.1 + // https://promisesaplus.com/#point-48 + if ( returned === deferred.promise() ) { + throw new TypeError( "Thenable self-resolution" ); + } + + // Support: Promises/A+ sections 2.3.3.1, 3.5 + // https://promisesaplus.com/#point-54 + // https://promisesaplus.com/#point-75 + // Retrieve `then` only once + then = returned && + + // Support: Promises/A+ section 2.3.4 + // https://promisesaplus.com/#point-64 + // Only check objects and functions for thenability + ( typeof returned === "object" || + typeof returned === "function" ) && + returned.then; + + // Handle a returned thenable + if ( isFunction( then ) ) { + + // Special processors (notify) just wait for resolution + if ( special ) { + then.call( + returned, + resolve( maxDepth, deferred, Identity, special ), + resolve( maxDepth, deferred, Thrower, special ) + ); + + // Normal processors (resolve) also hook into progress + } else { + + // ...and disregard older resolution values + maxDepth++; + + then.call( + returned, + resolve( maxDepth, deferred, Identity, special ), + resolve( maxDepth, deferred, Thrower, special ), + resolve( maxDepth, deferred, Identity, + deferred.notifyWith ) + ); + } + + // Handle all other returned values + } else { + + // Only substitute handlers pass on context + // and multiple values (non-spec behavior) + if ( handler !== Identity ) { + that = undefined; + args = [ returned ]; + } + + // Process the value(s) + // Default process is resolve + ( special || deferred.resolveWith )( that, args ); + } + }, + + // Only normal processors (resolve) catch and reject exceptions + process = special ? + mightThrow : + function() { + try { + mightThrow(); + } catch ( e ) { + + if ( jQuery.Deferred.exceptionHook ) { + jQuery.Deferred.exceptionHook( e, + process.stackTrace ); + } + + // Support: Promises/A+ section 2.3.3.3.4.1 + // https://promisesaplus.com/#point-61 + // Ignore post-resolution exceptions + if ( depth + 1 >= maxDepth ) { + + // Only substitute handlers pass on context + // and multiple values (non-spec behavior) + if ( handler !== Thrower ) { + that = undefined; + args = [ e ]; + } + + deferred.rejectWith( that, args ); + } + } + }; + + // Support: Promises/A+ section 2.3.3.3.1 + // https://promisesaplus.com/#point-57 + // Re-resolve promises immediately to dodge false rejection from + // subsequent errors + if ( depth ) { + process(); + } else { + + // Call an optional hook to record the stack, in case of exception + // since it's otherwise lost when execution goes async + if ( jQuery.Deferred.getStackHook ) { + process.stackTrace = jQuery.Deferred.getStackHook(); + } + window.setTimeout( process ); + } + }; + } + + return jQuery.Deferred( function( newDefer ) { + + // progress_handlers.add( ... ) + tuples[ 0 ][ 3 ].add( + resolve( + 0, + newDefer, + isFunction( onProgress ) ? + onProgress : + Identity, + newDefer.notifyWith + ) + ); + + // fulfilled_handlers.add( ... ) + tuples[ 1 ][ 3 ].add( + resolve( + 0, + newDefer, + isFunction( onFulfilled ) ? + onFulfilled : + Identity + ) + ); + + // rejected_handlers.add( ... ) + tuples[ 2 ][ 3 ].add( + resolve( + 0, + newDefer, + isFunction( onRejected ) ? + onRejected : + Thrower + ) + ); + } ).promise(); + }, + + // Get a promise for this deferred + // If obj is provided, the promise aspect is added to the object + promise: function( obj ) { + return obj != null ? jQuery.extend( obj, promise ) : promise; + } + }, + deferred = {}; + + // Add list-specific methods + jQuery.each( tuples, function( i, tuple ) { + var list = tuple[ 2 ], + stateString = tuple[ 5 ]; + + // promise.progress = list.add + // promise.done = list.add + // promise.fail = list.add + promise[ tuple[ 1 ] ] = list.add; + + // Handle state + if ( stateString ) { + list.add( + function() { + + // state = "resolved" (i.e., fulfilled) + // state = "rejected" + state = stateString; + }, + + // rejected_callbacks.disable + // fulfilled_callbacks.disable + tuples[ 3 - i ][ 2 ].disable, + + // rejected_handlers.disable + // fulfilled_handlers.disable + tuples[ 3 - i ][ 3 ].disable, + + // progress_callbacks.lock + tuples[ 0 ][ 2 ].lock, + + // progress_handlers.lock + tuples[ 0 ][ 3 ].lock + ); + } + + // progress_handlers.fire + // fulfilled_handlers.fire + // rejected_handlers.fire + list.add( tuple[ 3 ].fire ); + + // deferred.notify = function() { deferred.notifyWith(...) } + // deferred.resolve = function() { deferred.resolveWith(...) } + // deferred.reject = function() { deferred.rejectWith(...) } + deferred[ tuple[ 0 ] ] = function() { + deferred[ tuple[ 0 ] + "With" ]( this === deferred ? undefined : this, arguments ); + return this; + }; + + // deferred.notifyWith = list.fireWith + // deferred.resolveWith = list.fireWith + // deferred.rejectWith = list.fireWith + deferred[ tuple[ 0 ] + "With" ] = list.fireWith; + } ); + + // Make the deferred a promise + promise.promise( deferred ); + + // Call given func if any + if ( func ) { + func.call( deferred, deferred ); + } + + // All done! + return deferred; + }, + + // Deferred helper + when: function( singleValue ) { + var + + // count of uncompleted subordinates + remaining = arguments.length, + + // count of unprocessed arguments + i = remaining, + + // subordinate fulfillment data + resolveContexts = Array( i ), + resolveValues = slice.call( arguments ), + + // the master Deferred + master = jQuery.Deferred(), + + // subordinate callback factory + updateFunc = function( i ) { + return function( value ) { + resolveContexts[ i ] = this; + resolveValues[ i ] = arguments.length > 1 ? slice.call( arguments ) : value; + if ( !( --remaining ) ) { + master.resolveWith( resolveContexts, resolveValues ); + } + }; + }; + + // Single- and empty arguments are adopted like Promise.resolve + if ( remaining <= 1 ) { + adoptValue( singleValue, master.done( updateFunc( i ) ).resolve, master.reject, + !remaining ); + + // Use .then() to unwrap secondary thenables (cf. gh-3000) + if ( master.state() === "pending" || + isFunction( resolveValues[ i ] && resolveValues[ i ].then ) ) { + + return master.then(); + } + } + + // Multiple arguments are aggregated like Promise.all array elements + while ( i-- ) { + adoptValue( resolveValues[ i ], updateFunc( i ), master.reject ); + } + + return master.promise(); + } +} ); + + +// These usually indicate a programmer mistake during development, +// warn about them ASAP rather than swallowing them by default. +var rerrorNames = /^(Eval|Internal|Range|Reference|Syntax|Type|URI)Error$/; + +jQuery.Deferred.exceptionHook = function( error, stack ) { + + // Support: IE 8 - 9 only + // Console exists when dev tools are open, which can happen at any time + if ( window.console && window.console.warn && error && rerrorNames.test( error.name ) ) { + window.console.warn( "jQuery.Deferred exception: " + error.message, error.stack, stack ); + } +}; + + + + +jQuery.readyException = function( error ) { + window.setTimeout( function() { + throw error; + } ); +}; + + + + +// The deferred used on DOM ready +var readyList = jQuery.Deferred(); + +jQuery.fn.ready = function( fn ) { + + readyList + .then( fn ) + + // Wrap jQuery.readyException in a function so that the lookup + // happens at the time of error handling instead of callback + // registration. + .catch( function( error ) { + jQuery.readyException( error ); + } ); + + return this; +}; + +jQuery.extend( { + + // Is the DOM ready to be used? Set to true once it occurs. + isReady: false, + + // A counter to track how many items to wait for before + // the ready event fires. See #6781 + readyWait: 1, + + // Handle when the DOM is ready + ready: function( wait ) { + + // Abort if there are pending holds or we're already ready + if ( wait === true ? --jQuery.readyWait : jQuery.isReady ) { + return; + } + + // Remember that the DOM is ready + jQuery.isReady = true; + + // If a normal DOM Ready event fired, decrement, and wait if need be + if ( wait !== true && --jQuery.readyWait > 0 ) { + return; + } + + // If there are functions bound, to execute + readyList.resolveWith( document, [ jQuery ] ); + } +} ); + +jQuery.ready.then = readyList.then; + +// The ready event handler and self cleanup method +function completed() { + document.removeEventListener( "DOMContentLoaded", completed ); + window.removeEventListener( "load", completed ); + jQuery.ready(); +} + +// Catch cases where $(document).ready() is called +// after the browser event has already occurred. +// Support: IE <=9 - 10 only +// Older IE sometimes signals "interactive" too soon +if ( document.readyState === "complete" || + ( document.readyState !== "loading" && !document.documentElement.doScroll ) ) { + + // Handle it asynchronously to allow scripts the opportunity to delay ready + window.setTimeout( jQuery.ready ); + +} else { + + // Use the handy event callback + document.addEventListener( "DOMContentLoaded", completed ); + + // A fallback to window.onload, that will always work + window.addEventListener( "load", completed ); +} + + + + +// Multifunctional method to get and set values of a collection +// The value/s can optionally be executed if it's a function +var access = function( elems, fn, key, value, chainable, emptyGet, raw ) { + var i = 0, + len = elems.length, + bulk = key == null; + + // Sets many values + if ( toType( key ) === "object" ) { + chainable = true; + for ( i in key ) { + access( elems, fn, i, key[ i ], true, emptyGet, raw ); + } + + // Sets one value + } else if ( value !== undefined ) { + chainable = true; + + if ( !isFunction( value ) ) { + raw = true; + } + + if ( bulk ) { + + // Bulk operations run against the entire set + if ( raw ) { + fn.call( elems, value ); + fn = null; + + // ...except when executing function values + } else { + bulk = fn; + fn = function( elem, _key, value ) { + return bulk.call( jQuery( elem ), value ); + }; + } + } + + if ( fn ) { + for ( ; i < len; i++ ) { + fn( + elems[ i ], key, raw ? + value : + value.call( elems[ i ], i, fn( elems[ i ], key ) ) + ); + } + } + } + + if ( chainable ) { + return elems; + } + + // Gets + if ( bulk ) { + return fn.call( elems ); + } + + return len ? fn( elems[ 0 ], key ) : emptyGet; +}; + + +// Matches dashed string for camelizing +var rmsPrefix = /^-ms-/, + rdashAlpha = /-([a-z])/g; + +// Used by camelCase as callback to replace() +function fcamelCase( _all, letter ) { + return letter.toUpperCase(); +} + +// Convert dashed to camelCase; used by the css and data modules +// Support: IE <=9 - 11, Edge 12 - 15 +// Microsoft forgot to hump their vendor prefix (#9572) +function camelCase( string ) { + return string.replace( rmsPrefix, "ms-" ).replace( rdashAlpha, fcamelCase ); +} +var acceptData = function( owner ) { + + // Accepts only: + // - Node + // - Node.ELEMENT_NODE + // - Node.DOCUMENT_NODE + // - Object + // - Any + return owner.nodeType === 1 || owner.nodeType === 9 || !( +owner.nodeType ); +}; + + + + +function Data() { + this.expando = jQuery.expando + Data.uid++; +} + +Data.uid = 1; + +Data.prototype = { + + cache: function( owner ) { + + // Check if the owner object already has a cache + var value = owner[ this.expando ]; + + // If not, create one + if ( !value ) { + value = {}; + + // We can accept data for non-element nodes in modern browsers, + // but we should not, see #8335. + // Always return an empty object. + if ( acceptData( owner ) ) { + + // If it is a node unlikely to be stringify-ed or looped over + // use plain assignment + if ( owner.nodeType ) { + owner[ this.expando ] = value; + + // Otherwise secure it in a non-enumerable property + // configurable must be true to allow the property to be + // deleted when data is removed + } else { + Object.defineProperty( owner, this.expando, { + value: value, + configurable: true + } ); + } + } + } + + return value; + }, + set: function( owner, data, value ) { + var prop, + cache = this.cache( owner ); + + // Handle: [ owner, key, value ] args + // Always use camelCase key (gh-2257) + if ( typeof data === "string" ) { + cache[ camelCase( data ) ] = value; + + // Handle: [ owner, { properties } ] args + } else { + + // Copy the properties one-by-one to the cache object + for ( prop in data ) { + cache[ camelCase( prop ) ] = data[ prop ]; + } + } + return cache; + }, + get: function( owner, key ) { + return key === undefined ? + this.cache( owner ) : + + // Always use camelCase key (gh-2257) + owner[ this.expando ] && owner[ this.expando ][ camelCase( key ) ]; + }, + access: function( owner, key, value ) { + + // In cases where either: + // + // 1. No key was specified + // 2. A string key was specified, but no value provided + // + // Take the "read" path and allow the get method to determine + // which value to return, respectively either: + // + // 1. The entire cache object + // 2. The data stored at the key + // + if ( key === undefined || + ( ( key && typeof key === "string" ) && value === undefined ) ) { + + return this.get( owner, key ); + } + + // When the key is not a string, or both a key and value + // are specified, set or extend (existing objects) with either: + // + // 1. An object of properties + // 2. A key and value + // + this.set( owner, key, value ); + + // Since the "set" path can have two possible entry points + // return the expected data based on which path was taken[*] + return value !== undefined ? value : key; + }, + remove: function( owner, key ) { + var i, + cache = owner[ this.expando ]; + + if ( cache === undefined ) { + return; + } + + if ( key !== undefined ) { + + // Support array or space separated string of keys + if ( Array.isArray( key ) ) { + + // If key is an array of keys... + // We always set camelCase keys, so remove that. + key = key.map( camelCase ); + } else { + key = camelCase( key ); + + // If a key with the spaces exists, use it. + // Otherwise, create an array by matching non-whitespace + key = key in cache ? + [ key ] : + ( key.match( rnothtmlwhite ) || [] ); + } + + i = key.length; + + while ( i-- ) { + delete cache[ key[ i ] ]; + } + } + + // Remove the expando if there's no more data + if ( key === undefined || jQuery.isEmptyObject( cache ) ) { + + // Support: Chrome <=35 - 45 + // Webkit & Blink performance suffers when deleting properties + // from DOM nodes, so set to undefined instead + // https://bugs.chromium.org/p/chromium/issues/detail?id=378607 (bug restricted) + if ( owner.nodeType ) { + owner[ this.expando ] = undefined; + } else { + delete owner[ this.expando ]; + } + } + }, + hasData: function( owner ) { + var cache = owner[ this.expando ]; + return cache !== undefined && !jQuery.isEmptyObject( cache ); + } +}; +var dataPriv = new Data(); + +var dataUser = new Data(); + + + +// Implementation Summary +// +// 1. Enforce API surface and semantic compatibility with 1.9.x branch +// 2. Improve the module's maintainability by reducing the storage +// paths to a single mechanism. +// 3. Use the same single mechanism to support "private" and "user" data. +// 4. _Never_ expose "private" data to user code (TODO: Drop _data, _removeData) +// 5. Avoid exposing implementation details on user objects (eg. expando properties) +// 6. Provide a clear path for implementation upgrade to WeakMap in 2014 + +var rbrace = /^(?:\{[\w\W]*\}|\[[\w\W]*\])$/, + rmultiDash = /[A-Z]/g; + +function getData( data ) { + if ( data === "true" ) { + return true; + } + + if ( data === "false" ) { + return false; + } + + if ( data === "null" ) { + return null; + } + + // Only convert to a number if it doesn't change the string + if ( data === +data + "" ) { + return +data; + } + + if ( rbrace.test( data ) ) { + return JSON.parse( data ); + } + + return data; +} + +function dataAttr( elem, key, data ) { + var name; + + // If nothing was found internally, try to fetch any + // data from the HTML5 data-* attribute + if ( data === undefined && elem.nodeType === 1 ) { + name = "data-" + key.replace( rmultiDash, "-$&" ).toLowerCase(); + data = elem.getAttribute( name ); + + if ( typeof data === "string" ) { + try { + data = getData( data ); + } catch ( e ) {} + + // Make sure we set the data so it isn't changed later + dataUser.set( elem, key, data ); + } else { + data = undefined; + } + } + return data; +} + +jQuery.extend( { + hasData: function( elem ) { + return dataUser.hasData( elem ) || dataPriv.hasData( elem ); + }, + + data: function( elem, name, data ) { + return dataUser.access( elem, name, data ); + }, + + removeData: function( elem, name ) { + dataUser.remove( elem, name ); + }, + + // TODO: Now that all calls to _data and _removeData have been replaced + // with direct calls to dataPriv methods, these can be deprecated. + _data: function( elem, name, data ) { + return dataPriv.access( elem, name, data ); + }, + + _removeData: function( elem, name ) { + dataPriv.remove( elem, name ); + } +} ); + +jQuery.fn.extend( { + data: function( key, value ) { + var i, name, data, + elem = this[ 0 ], + attrs = elem && elem.attributes; + + // Gets all values + if ( key === undefined ) { + if ( this.length ) { + data = dataUser.get( elem ); + + if ( elem.nodeType === 1 && !dataPriv.get( elem, "hasDataAttrs" ) ) { + i = attrs.length; + while ( i-- ) { + + // Support: IE 11 only + // The attrs elements can be null (#14894) + if ( attrs[ i ] ) { + name = attrs[ i ].name; + if ( name.indexOf( "data-" ) === 0 ) { + name = camelCase( name.slice( 5 ) ); + dataAttr( elem, name, data[ name ] ); + } + } + } + dataPriv.set( elem, "hasDataAttrs", true ); + } + } + + return data; + } + + // Sets multiple values + if ( typeof key === "object" ) { + return this.each( function() { + dataUser.set( this, key ); + } ); + } + + return access( this, function( value ) { + var data; + + // The calling jQuery object (element matches) is not empty + // (and therefore has an element appears at this[ 0 ]) and the + // `value` parameter was not undefined. An empty jQuery object + // will result in `undefined` for elem = this[ 0 ] which will + // throw an exception if an attempt to read a data cache is made. + if ( elem && value === undefined ) { + + // Attempt to get data from the cache + // The key will always be camelCased in Data + data = dataUser.get( elem, key ); + if ( data !== undefined ) { + return data; + } + + // Attempt to "discover" the data in + // HTML5 custom data-* attrs + data = dataAttr( elem, key ); + if ( data !== undefined ) { + return data; + } + + // We tried really hard, but the data doesn't exist. + return; + } + + // Set the data... + this.each( function() { + + // We always store the camelCased key + dataUser.set( this, key, value ); + } ); + }, null, value, arguments.length > 1, null, true ); + }, + + removeData: function( key ) { + return this.each( function() { + dataUser.remove( this, key ); + } ); + } +} ); + + +jQuery.extend( { + queue: function( elem, type, data ) { + var queue; + + if ( elem ) { + type = ( type || "fx" ) + "queue"; + queue = dataPriv.get( elem, type ); + + // Speed up dequeue by getting out quickly if this is just a lookup + if ( data ) { + if ( !queue || Array.isArray( data ) ) { + queue = dataPriv.access( elem, type, jQuery.makeArray( data ) ); + } else { + queue.push( data ); + } + } + return queue || []; + } + }, + + dequeue: function( elem, type ) { + type = type || "fx"; + + var queue = jQuery.queue( elem, type ), + startLength = queue.length, + fn = queue.shift(), + hooks = jQuery._queueHooks( elem, type ), + next = function() { + jQuery.dequeue( elem, type ); + }; + + // If the fx queue is dequeued, always remove the progress sentinel + if ( fn === "inprogress" ) { + fn = queue.shift(); + startLength--; + } + + if ( fn ) { + + // Add a progress sentinel to prevent the fx queue from being + // automatically dequeued + if ( type === "fx" ) { + queue.unshift( "inprogress" ); + } + + // Clear up the last queue stop function + delete hooks.stop; + fn.call( elem, next, hooks ); + } + + if ( !startLength && hooks ) { + hooks.empty.fire(); + } + }, + + // Not public - generate a queueHooks object, or return the current one + _queueHooks: function( elem, type ) { + var key = type + "queueHooks"; + return dataPriv.get( elem, key ) || dataPriv.access( elem, key, { + empty: jQuery.Callbacks( "once memory" ).add( function() { + dataPriv.remove( elem, [ type + "queue", key ] ); + } ) + } ); + } +} ); + +jQuery.fn.extend( { + queue: function( type, data ) { + var setter = 2; + + if ( typeof type !== "string" ) { + data = type; + type = "fx"; + setter--; + } + + if ( arguments.length < setter ) { + return jQuery.queue( this[ 0 ], type ); + } + + return data === undefined ? + this : + this.each( function() { + var queue = jQuery.queue( this, type, data ); + + // Ensure a hooks for this queue + jQuery._queueHooks( this, type ); + + if ( type === "fx" && queue[ 0 ] !== "inprogress" ) { + jQuery.dequeue( this, type ); + } + } ); + }, + dequeue: function( type ) { + return this.each( function() { + jQuery.dequeue( this, type ); + } ); + }, + clearQueue: function( type ) { + return this.queue( type || "fx", [] ); + }, + + // Get a promise resolved when queues of a certain type + // are emptied (fx is the type by default) + promise: function( type, obj ) { + var tmp, + count = 1, + defer = jQuery.Deferred(), + elements = this, + i = this.length, + resolve = function() { + if ( !( --count ) ) { + defer.resolveWith( elements, [ elements ] ); + } + }; + + if ( typeof type !== "string" ) { + obj = type; + type = undefined; + } + type = type || "fx"; + + while ( i-- ) { + tmp = dataPriv.get( elements[ i ], type + "queueHooks" ); + if ( tmp && tmp.empty ) { + count++; + tmp.empty.add( resolve ); + } + } + resolve(); + return defer.promise( obj ); + } +} ); +var pnum = ( /[+-]?(?:\d*\.|)\d+(?:[eE][+-]?\d+|)/ ).source; + +var rcssNum = new RegExp( "^(?:([+-])=|)(" + pnum + ")([a-z%]*)$", "i" ); + + +var cssExpand = [ "Top", "Right", "Bottom", "Left" ]; + +var documentElement = document.documentElement; + + + + var isAttached = function( elem ) { + return jQuery.contains( elem.ownerDocument, elem ); + }, + composed = { composed: true }; + + // Support: IE 9 - 11+, Edge 12 - 18+, iOS 10.0 - 10.2 only + // Check attachment across shadow DOM boundaries when possible (gh-3504) + // Support: iOS 10.0-10.2 only + // Early iOS 10 versions support `attachShadow` but not `getRootNode`, + // leading to errors. We need to check for `getRootNode`. + if ( documentElement.getRootNode ) { + isAttached = function( elem ) { + return jQuery.contains( elem.ownerDocument, elem ) || + elem.getRootNode( composed ) === elem.ownerDocument; + }; + } +var isHiddenWithinTree = function( elem, el ) { + + // isHiddenWithinTree might be called from jQuery#filter function; + // in that case, element will be second argument + elem = el || elem; + + // Inline style trumps all + return elem.style.display === "none" || + elem.style.display === "" && + + // Otherwise, check computed style + // Support: Firefox <=43 - 45 + // Disconnected elements can have computed display: none, so first confirm that elem is + // in the document. + isAttached( elem ) && + + jQuery.css( elem, "display" ) === "none"; + }; + + + +function adjustCSS( elem, prop, valueParts, tween ) { + var adjusted, scale, + maxIterations = 20, + currentValue = tween ? + function() { + return tween.cur(); + } : + function() { + return jQuery.css( elem, prop, "" ); + }, + initial = currentValue(), + unit = valueParts && valueParts[ 3 ] || ( jQuery.cssNumber[ prop ] ? "" : "px" ), + + // Starting value computation is required for potential unit mismatches + initialInUnit = elem.nodeType && + ( jQuery.cssNumber[ prop ] || unit !== "px" && +initial ) && + rcssNum.exec( jQuery.css( elem, prop ) ); + + if ( initialInUnit && initialInUnit[ 3 ] !== unit ) { + + // Support: Firefox <=54 + // Halve the iteration target value to prevent interference from CSS upper bounds (gh-2144) + initial = initial / 2; + + // Trust units reported by jQuery.css + unit = unit || initialInUnit[ 3 ]; + + // Iteratively approximate from a nonzero starting point + initialInUnit = +initial || 1; + + while ( maxIterations-- ) { + + // Evaluate and update our best guess (doubling guesses that zero out). + // Finish if the scale equals or crosses 1 (making the old*new product non-positive). + jQuery.style( elem, prop, initialInUnit + unit ); + if ( ( 1 - scale ) * ( 1 - ( scale = currentValue() / initial || 0.5 ) ) <= 0 ) { + maxIterations = 0; + } + initialInUnit = initialInUnit / scale; + + } + + initialInUnit = initialInUnit * 2; + jQuery.style( elem, prop, initialInUnit + unit ); + + // Make sure we update the tween properties later on + valueParts = valueParts || []; + } + + if ( valueParts ) { + initialInUnit = +initialInUnit || +initial || 0; + + // Apply relative offset (+=/-=) if specified + adjusted = valueParts[ 1 ] ? + initialInUnit + ( valueParts[ 1 ] + 1 ) * valueParts[ 2 ] : + +valueParts[ 2 ]; + if ( tween ) { + tween.unit = unit; + tween.start = initialInUnit; + tween.end = adjusted; + } + } + return adjusted; +} + + +var defaultDisplayMap = {}; + +function getDefaultDisplay( elem ) { + var temp, + doc = elem.ownerDocument, + nodeName = elem.nodeName, + display = defaultDisplayMap[ nodeName ]; + + if ( display ) { + return display; + } + + temp = doc.body.appendChild( doc.createElement( nodeName ) ); + display = jQuery.css( temp, "display" ); + + temp.parentNode.removeChild( temp ); + + if ( display === "none" ) { + display = "block"; + } + defaultDisplayMap[ nodeName ] = display; + + return display; +} + +function showHide( elements, show ) { + var display, elem, + values = [], + index = 0, + length = elements.length; + + // Determine new display value for elements that need to change + for ( ; index < length; index++ ) { + elem = elements[ index ]; + if ( !elem.style ) { + continue; + } + + display = elem.style.display; + if ( show ) { + + // Since we force visibility upon cascade-hidden elements, an immediate (and slow) + // check is required in this first loop unless we have a nonempty display value (either + // inline or about-to-be-restored) + if ( display === "none" ) { + values[ index ] = dataPriv.get( elem, "display" ) || null; + if ( !values[ index ] ) { + elem.style.display = ""; + } + } + if ( elem.style.display === "" && isHiddenWithinTree( elem ) ) { + values[ index ] = getDefaultDisplay( elem ); + } + } else { + if ( display !== "none" ) { + values[ index ] = "none"; + + // Remember what we're overwriting + dataPriv.set( elem, "display", display ); + } + } + } + + // Set the display of the elements in a second loop to avoid constant reflow + for ( index = 0; index < length; index++ ) { + if ( values[ index ] != null ) { + elements[ index ].style.display = values[ index ]; + } + } + + return elements; +} + +jQuery.fn.extend( { + show: function() { + return showHide( this, true ); + }, + hide: function() { + return showHide( this ); + }, + toggle: function( state ) { + if ( typeof state === "boolean" ) { + return state ? this.show() : this.hide(); + } + + return this.each( function() { + if ( isHiddenWithinTree( this ) ) { + jQuery( this ).show(); + } else { + jQuery( this ).hide(); + } + } ); + } +} ); +var rcheckableType = ( /^(?:checkbox|radio)$/i ); + +var rtagName = ( /<([a-z][^\/\0>\x20\t\r\n\f]*)/i ); + +var rscriptType = ( /^$|^module$|\/(?:java|ecma)script/i ); + + + +( function() { + var fragment = document.createDocumentFragment(), + div = fragment.appendChild( document.createElement( "div" ) ), + input = document.createElement( "input" ); + + // Support: Android 4.0 - 4.3 only + // Check state lost if the name is set (#11217) + // Support: Windows Web Apps (WWA) + // `name` and `type` must use .setAttribute for WWA (#14901) + input.setAttribute( "type", "radio" ); + input.setAttribute( "checked", "checked" ); + input.setAttribute( "name", "t" ); + + div.appendChild( input ); + + // Support: Android <=4.1 only + // Older WebKit doesn't clone checked state correctly in fragments + support.checkClone = div.cloneNode( true ).cloneNode( true ).lastChild.checked; + + // Support: IE <=11 only + // Make sure textarea (and checkbox) defaultValue is properly cloned + div.innerHTML = ""; + support.noCloneChecked = !!div.cloneNode( true ).lastChild.defaultValue; + + // Support: IE <=9 only + // IE <=9 replaces "; + support.option = !!div.lastChild; +} )(); + + +// We have to close these tags to support XHTML (#13200) +var wrapMap = { + + // XHTML parsers do not magically insert elements in the + // same way that tag soup parsers do. So we cannot shorten + // this by omitting or other required elements. + thead: [ 1, "", "
" ], + col: [ 2, "", "
" ], + tr: [ 2, "", "
" ], + td: [ 3, "", "
" ], + + _default: [ 0, "", "" ] +}; + +wrapMap.tbody = wrapMap.tfoot = wrapMap.colgroup = wrapMap.caption = wrapMap.thead; +wrapMap.th = wrapMap.td; + +// Support: IE <=9 only +if ( !support.option ) { + wrapMap.optgroup = wrapMap.option = [ 1, "" ]; +} + + +function getAll( context, tag ) { + + // Support: IE <=9 - 11 only + // Use typeof to avoid zero-argument method invocation on host objects (#15151) + var ret; + + if ( typeof context.getElementsByTagName !== "undefined" ) { + ret = context.getElementsByTagName( tag || "*" ); + + } else if ( typeof context.querySelectorAll !== "undefined" ) { + ret = context.querySelectorAll( tag || "*" ); + + } else { + ret = []; + } + + if ( tag === undefined || tag && nodeName( context, tag ) ) { + return jQuery.merge( [ context ], ret ); + } + + return ret; +} + + +// Mark scripts as having already been evaluated +function setGlobalEval( elems, refElements ) { + var i = 0, + l = elems.length; + + for ( ; i < l; i++ ) { + dataPriv.set( + elems[ i ], + "globalEval", + !refElements || dataPriv.get( refElements[ i ], "globalEval" ) + ); + } +} + + +var rhtml = /<|&#?\w+;/; + +function buildFragment( elems, context, scripts, selection, ignored ) { + var elem, tmp, tag, wrap, attached, j, + fragment = context.createDocumentFragment(), + nodes = [], + i = 0, + l = elems.length; + + for ( ; i < l; i++ ) { + elem = elems[ i ]; + + if ( elem || elem === 0 ) { + + // Add nodes directly + if ( toType( elem ) === "object" ) { + + // Support: Android <=4.0 only, PhantomJS 1 only + // push.apply(_, arraylike) throws on ancient WebKit + jQuery.merge( nodes, elem.nodeType ? [ elem ] : elem ); + + // Convert non-html into a text node + } else if ( !rhtml.test( elem ) ) { + nodes.push( context.createTextNode( elem ) ); + + // Convert html into DOM nodes + } else { + tmp = tmp || fragment.appendChild( context.createElement( "div" ) ); + + // Deserialize a standard representation + tag = ( rtagName.exec( elem ) || [ "", "" ] )[ 1 ].toLowerCase(); + wrap = wrapMap[ tag ] || wrapMap._default; + tmp.innerHTML = wrap[ 1 ] + jQuery.htmlPrefilter( elem ) + wrap[ 2 ]; + + // Descend through wrappers to the right content + j = wrap[ 0 ]; + while ( j-- ) { + tmp = tmp.lastChild; + } + + // Support: Android <=4.0 only, PhantomJS 1 only + // push.apply(_, arraylike) throws on ancient WebKit + jQuery.merge( nodes, tmp.childNodes ); + + // Remember the top-level container + tmp = fragment.firstChild; + + // Ensure the created nodes are orphaned (#12392) + tmp.textContent = ""; + } + } + } + + // Remove wrapper from fragment + fragment.textContent = ""; + + i = 0; + while ( ( elem = nodes[ i++ ] ) ) { + + // Skip elements already in the context collection (trac-4087) + if ( selection && jQuery.inArray( elem, selection ) > -1 ) { + if ( ignored ) { + ignored.push( elem ); + } + continue; + } + + attached = isAttached( elem ); + + // Append to fragment + tmp = getAll( fragment.appendChild( elem ), "script" ); + + // Preserve script evaluation history + if ( attached ) { + setGlobalEval( tmp ); + } + + // Capture executables + if ( scripts ) { + j = 0; + while ( ( elem = tmp[ j++ ] ) ) { + if ( rscriptType.test( elem.type || "" ) ) { + scripts.push( elem ); + } + } + } + } + + return fragment; +} + + +var + rkeyEvent = /^key/, + rmouseEvent = /^(?:mouse|pointer|contextmenu|drag|drop)|click/, + rtypenamespace = /^([^.]*)(?:\.(.+)|)/; + +function returnTrue() { + return true; +} + +function returnFalse() { + return false; +} + +// Support: IE <=9 - 11+ +// focus() and blur() are asynchronous, except when they are no-op. +// So expect focus to be synchronous when the element is already active, +// and blur to be synchronous when the element is not already active. +// (focus and blur are always synchronous in other supported browsers, +// this just defines when we can count on it). +function expectSync( elem, type ) { + return ( elem === safeActiveElement() ) === ( type === "focus" ); +} + +// Support: IE <=9 only +// Accessing document.activeElement can throw unexpectedly +// https://bugs.jquery.com/ticket/13393 +function safeActiveElement() { + try { + return document.activeElement; + } catch ( err ) { } +} + +function on( elem, types, selector, data, fn, one ) { + var origFn, type; + + // Types can be a map of types/handlers + if ( typeof types === "object" ) { + + // ( types-Object, selector, data ) + if ( typeof selector !== "string" ) { + + // ( types-Object, data ) + data = data || selector; + selector = undefined; + } + for ( type in types ) { + on( elem, type, selector, data, types[ type ], one ); + } + return elem; + } + + if ( data == null && fn == null ) { + + // ( types, fn ) + fn = selector; + data = selector = undefined; + } else if ( fn == null ) { + if ( typeof selector === "string" ) { + + // ( types, selector, fn ) + fn = data; + data = undefined; + } else { + + // ( types, data, fn ) + fn = data; + data = selector; + selector = undefined; + } + } + if ( fn === false ) { + fn = returnFalse; + } else if ( !fn ) { + return elem; + } + + if ( one === 1 ) { + origFn = fn; + fn = function( event ) { + + // Can use an empty set, since event contains the info + jQuery().off( event ); + return origFn.apply( this, arguments ); + }; + + // Use same guid so caller can remove using origFn + fn.guid = origFn.guid || ( origFn.guid = jQuery.guid++ ); + } + return elem.each( function() { + jQuery.event.add( this, types, fn, data, selector ); + } ); +} + +/* + * Helper functions for managing events -- not part of the public interface. + * Props to Dean Edwards' addEvent library for many of the ideas. + */ +jQuery.event = { + + global: {}, + + add: function( elem, types, handler, data, selector ) { + + var handleObjIn, eventHandle, tmp, + events, t, handleObj, + special, handlers, type, namespaces, origType, + elemData = dataPriv.get( elem ); + + // Only attach events to objects that accept data + if ( !acceptData( elem ) ) { + return; + } + + // Caller can pass in an object of custom data in lieu of the handler + if ( handler.handler ) { + handleObjIn = handler; + handler = handleObjIn.handler; + selector = handleObjIn.selector; + } + + // Ensure that invalid selectors throw exceptions at attach time + // Evaluate against documentElement in case elem is a non-element node (e.g., document) + if ( selector ) { + jQuery.find.matchesSelector( documentElement, selector ); + } + + // Make sure that the handler has a unique ID, used to find/remove it later + if ( !handler.guid ) { + handler.guid = jQuery.guid++; + } + + // Init the element's event structure and main handler, if this is the first + if ( !( events = elemData.events ) ) { + events = elemData.events = Object.create( null ); + } + if ( !( eventHandle = elemData.handle ) ) { + eventHandle = elemData.handle = function( e ) { + + // Discard the second event of a jQuery.event.trigger() and + // when an event is called after a page has unloaded + return typeof jQuery !== "undefined" && jQuery.event.triggered !== e.type ? + jQuery.event.dispatch.apply( elem, arguments ) : undefined; + }; + } + + // Handle multiple events separated by a space + types = ( types || "" ).match( rnothtmlwhite ) || [ "" ]; + t = types.length; + while ( t-- ) { + tmp = rtypenamespace.exec( types[ t ] ) || []; + type = origType = tmp[ 1 ]; + namespaces = ( tmp[ 2 ] || "" ).split( "." ).sort(); + + // There *must* be a type, no attaching namespace-only handlers + if ( !type ) { + continue; + } + + // If event changes its type, use the special event handlers for the changed type + special = jQuery.event.special[ type ] || {}; + + // If selector defined, determine special event api type, otherwise given type + type = ( selector ? special.delegateType : special.bindType ) || type; + + // Update special based on newly reset type + special = jQuery.event.special[ type ] || {}; + + // handleObj is passed to all event handlers + handleObj = jQuery.extend( { + type: type, + origType: origType, + data: data, + handler: handler, + guid: handler.guid, + selector: selector, + needsContext: selector && jQuery.expr.match.needsContext.test( selector ), + namespace: namespaces.join( "." ) + }, handleObjIn ); + + // Init the event handler queue if we're the first + if ( !( handlers = events[ type ] ) ) { + handlers = events[ type ] = []; + handlers.delegateCount = 0; + + // Only use addEventListener if the special events handler returns false + if ( !special.setup || + special.setup.call( elem, data, namespaces, eventHandle ) === false ) { + + if ( elem.addEventListener ) { + elem.addEventListener( type, eventHandle ); + } + } + } + + if ( special.add ) { + special.add.call( elem, handleObj ); + + if ( !handleObj.handler.guid ) { + handleObj.handler.guid = handler.guid; + } + } + + // Add to the element's handler list, delegates in front + if ( selector ) { + handlers.splice( handlers.delegateCount++, 0, handleObj ); + } else { + handlers.push( handleObj ); + } + + // Keep track of which events have ever been used, for event optimization + jQuery.event.global[ type ] = true; + } + + }, + + // Detach an event or set of events from an element + remove: function( elem, types, handler, selector, mappedTypes ) { + + var j, origCount, tmp, + events, t, handleObj, + special, handlers, type, namespaces, origType, + elemData = dataPriv.hasData( elem ) && dataPriv.get( elem ); + + if ( !elemData || !( events = elemData.events ) ) { + return; + } + + // Once for each type.namespace in types; type may be omitted + types = ( types || "" ).match( rnothtmlwhite ) || [ "" ]; + t = types.length; + while ( t-- ) { + tmp = rtypenamespace.exec( types[ t ] ) || []; + type = origType = tmp[ 1 ]; + namespaces = ( tmp[ 2 ] || "" ).split( "." ).sort(); + + // Unbind all events (on this namespace, if provided) for the element + if ( !type ) { + for ( type in events ) { + jQuery.event.remove( elem, type + types[ t ], handler, selector, true ); + } + continue; + } + + special = jQuery.event.special[ type ] || {}; + type = ( selector ? special.delegateType : special.bindType ) || type; + handlers = events[ type ] || []; + tmp = tmp[ 2 ] && + new RegExp( "(^|\\.)" + namespaces.join( "\\.(?:.*\\.|)" ) + "(\\.|$)" ); + + // Remove matching events + origCount = j = handlers.length; + while ( j-- ) { + handleObj = handlers[ j ]; + + if ( ( mappedTypes || origType === handleObj.origType ) && + ( !handler || handler.guid === handleObj.guid ) && + ( !tmp || tmp.test( handleObj.namespace ) ) && + ( !selector || selector === handleObj.selector || + selector === "**" && handleObj.selector ) ) { + handlers.splice( j, 1 ); + + if ( handleObj.selector ) { + handlers.delegateCount--; + } + if ( special.remove ) { + special.remove.call( elem, handleObj ); + } + } + } + + // Remove generic event handler if we removed something and no more handlers exist + // (avoids potential for endless recursion during removal of special event handlers) + if ( origCount && !handlers.length ) { + if ( !special.teardown || + special.teardown.call( elem, namespaces, elemData.handle ) === false ) { + + jQuery.removeEvent( elem, type, elemData.handle ); + } + + delete events[ type ]; + } + } + + // Remove data and the expando if it's no longer used + if ( jQuery.isEmptyObject( events ) ) { + dataPriv.remove( elem, "handle events" ); + } + }, + + dispatch: function( nativeEvent ) { + + var i, j, ret, matched, handleObj, handlerQueue, + args = new Array( arguments.length ), + + // Make a writable jQuery.Event from the native event object + event = jQuery.event.fix( nativeEvent ), + + handlers = ( + dataPriv.get( this, "events" ) || Object.create( null ) + )[ event.type ] || [], + special = jQuery.event.special[ event.type ] || {}; + + // Use the fix-ed jQuery.Event rather than the (read-only) native event + args[ 0 ] = event; + + for ( i = 1; i < arguments.length; i++ ) { + args[ i ] = arguments[ i ]; + } + + event.delegateTarget = this; + + // Call the preDispatch hook for the mapped type, and let it bail if desired + if ( special.preDispatch && special.preDispatch.call( this, event ) === false ) { + return; + } + + // Determine handlers + handlerQueue = jQuery.event.handlers.call( this, event, handlers ); + + // Run delegates first; they may want to stop propagation beneath us + i = 0; + while ( ( matched = handlerQueue[ i++ ] ) && !event.isPropagationStopped() ) { + event.currentTarget = matched.elem; + + j = 0; + while ( ( handleObj = matched.handlers[ j++ ] ) && + !event.isImmediatePropagationStopped() ) { + + // If the event is namespaced, then each handler is only invoked if it is + // specially universal or its namespaces are a superset of the event's. + if ( !event.rnamespace || handleObj.namespace === false || + event.rnamespace.test( handleObj.namespace ) ) { + + event.handleObj = handleObj; + event.data = handleObj.data; + + ret = ( ( jQuery.event.special[ handleObj.origType ] || {} ).handle || + handleObj.handler ).apply( matched.elem, args ); + + if ( ret !== undefined ) { + if ( ( event.result = ret ) === false ) { + event.preventDefault(); + event.stopPropagation(); + } + } + } + } + } + + // Call the postDispatch hook for the mapped type + if ( special.postDispatch ) { + special.postDispatch.call( this, event ); + } + + return event.result; + }, + + handlers: function( event, handlers ) { + var i, handleObj, sel, matchedHandlers, matchedSelectors, + handlerQueue = [], + delegateCount = handlers.delegateCount, + cur = event.target; + + // Find delegate handlers + if ( delegateCount && + + // Support: IE <=9 + // Black-hole SVG instance trees (trac-13180) + cur.nodeType && + + // Support: Firefox <=42 + // Suppress spec-violating clicks indicating a non-primary pointer button (trac-3861) + // https://www.w3.org/TR/DOM-Level-3-Events/#event-type-click + // Support: IE 11 only + // ...but not arrow key "clicks" of radio inputs, which can have `button` -1 (gh-2343) + !( event.type === "click" && event.button >= 1 ) ) { + + for ( ; cur !== this; cur = cur.parentNode || this ) { + + // Don't check non-elements (#13208) + // Don't process clicks on disabled elements (#6911, #8165, #11382, #11764) + if ( cur.nodeType === 1 && !( event.type === "click" && cur.disabled === true ) ) { + matchedHandlers = []; + matchedSelectors = {}; + for ( i = 0; i < delegateCount; i++ ) { + handleObj = handlers[ i ]; + + // Don't conflict with Object.prototype properties (#13203) + sel = handleObj.selector + " "; + + if ( matchedSelectors[ sel ] === undefined ) { + matchedSelectors[ sel ] = handleObj.needsContext ? + jQuery( sel, this ).index( cur ) > -1 : + jQuery.find( sel, this, null, [ cur ] ).length; + } + if ( matchedSelectors[ sel ] ) { + matchedHandlers.push( handleObj ); + } + } + if ( matchedHandlers.length ) { + handlerQueue.push( { elem: cur, handlers: matchedHandlers } ); + } + } + } + } + + // Add the remaining (directly-bound) handlers + cur = this; + if ( delegateCount < handlers.length ) { + handlerQueue.push( { elem: cur, handlers: handlers.slice( delegateCount ) } ); + } + + return handlerQueue; + }, + + addProp: function( name, hook ) { + Object.defineProperty( jQuery.Event.prototype, name, { + enumerable: true, + configurable: true, + + get: isFunction( hook ) ? + function() { + if ( this.originalEvent ) { + return hook( this.originalEvent ); + } + } : + function() { + if ( this.originalEvent ) { + return this.originalEvent[ name ]; + } + }, + + set: function( value ) { + Object.defineProperty( this, name, { + enumerable: true, + configurable: true, + writable: true, + value: value + } ); + } + } ); + }, + + fix: function( originalEvent ) { + return originalEvent[ jQuery.expando ] ? + originalEvent : + new jQuery.Event( originalEvent ); + }, + + special: { + load: { + + // Prevent triggered image.load events from bubbling to window.load + noBubble: true + }, + click: { + + // Utilize native event to ensure correct state for checkable inputs + setup: function( data ) { + + // For mutual compressibility with _default, replace `this` access with a local var. + // `|| data` is dead code meant only to preserve the variable through minification. + var el = this || data; + + // Claim the first handler + if ( rcheckableType.test( el.type ) && + el.click && nodeName( el, "input" ) ) { + + // dataPriv.set( el, "click", ... ) + leverageNative( el, "click", returnTrue ); + } + + // Return false to allow normal processing in the caller + return false; + }, + trigger: function( data ) { + + // For mutual compressibility with _default, replace `this` access with a local var. + // `|| data` is dead code meant only to preserve the variable through minification. + var el = this || data; + + // Force setup before triggering a click + if ( rcheckableType.test( el.type ) && + el.click && nodeName( el, "input" ) ) { + + leverageNative( el, "click" ); + } + + // Return non-false to allow normal event-path propagation + return true; + }, + + // For cross-browser consistency, suppress native .click() on links + // Also prevent it if we're currently inside a leveraged native-event stack + _default: function( event ) { + var target = event.target; + return rcheckableType.test( target.type ) && + target.click && nodeName( target, "input" ) && + dataPriv.get( target, "click" ) || + nodeName( target, "a" ); + } + }, + + beforeunload: { + postDispatch: function( event ) { + + // Support: Firefox 20+ + // Firefox doesn't alert if the returnValue field is not set. + if ( event.result !== undefined && event.originalEvent ) { + event.originalEvent.returnValue = event.result; + } + } + } + } +}; + +// Ensure the presence of an event listener that handles manually-triggered +// synthetic events by interrupting progress until reinvoked in response to +// *native* events that it fires directly, ensuring that state changes have +// already occurred before other listeners are invoked. +function leverageNative( el, type, expectSync ) { + + // Missing expectSync indicates a trigger call, which must force setup through jQuery.event.add + if ( !expectSync ) { + if ( dataPriv.get( el, type ) === undefined ) { + jQuery.event.add( el, type, returnTrue ); + } + return; + } + + // Register the controller as a special universal handler for all event namespaces + dataPriv.set( el, type, false ); + jQuery.event.add( el, type, { + namespace: false, + handler: function( event ) { + var notAsync, result, + saved = dataPriv.get( this, type ); + + if ( ( event.isTrigger & 1 ) && this[ type ] ) { + + // Interrupt processing of the outer synthetic .trigger()ed event + // Saved data should be false in such cases, but might be a leftover capture object + // from an async native handler (gh-4350) + if ( !saved.length ) { + + // Store arguments for use when handling the inner native event + // There will always be at least one argument (an event object), so this array + // will not be confused with a leftover capture object. + saved = slice.call( arguments ); + dataPriv.set( this, type, saved ); + + // Trigger the native event and capture its result + // Support: IE <=9 - 11+ + // focus() and blur() are asynchronous + notAsync = expectSync( this, type ); + this[ type ](); + result = dataPriv.get( this, type ); + if ( saved !== result || notAsync ) { + dataPriv.set( this, type, false ); + } else { + result = {}; + } + if ( saved !== result ) { + + // Cancel the outer synthetic event + event.stopImmediatePropagation(); + event.preventDefault(); + return result.value; + } + + // If this is an inner synthetic event for an event with a bubbling surrogate + // (focus or blur), assume that the surrogate already propagated from triggering the + // native event and prevent that from happening again here. + // This technically gets the ordering wrong w.r.t. to `.trigger()` (in which the + // bubbling surrogate propagates *after* the non-bubbling base), but that seems + // less bad than duplication. + } else if ( ( jQuery.event.special[ type ] || {} ).delegateType ) { + event.stopPropagation(); + } + + // If this is a native event triggered above, everything is now in order + // Fire an inner synthetic event with the original arguments + } else if ( saved.length ) { + + // ...and capture the result + dataPriv.set( this, type, { + value: jQuery.event.trigger( + + // Support: IE <=9 - 11+ + // Extend with the prototype to reset the above stopImmediatePropagation() + jQuery.extend( saved[ 0 ], jQuery.Event.prototype ), + saved.slice( 1 ), + this + ) + } ); + + // Abort handling of the native event + event.stopImmediatePropagation(); + } + } + } ); +} + +jQuery.removeEvent = function( elem, type, handle ) { + + // This "if" is needed for plain objects + if ( elem.removeEventListener ) { + elem.removeEventListener( type, handle ); + } +}; + +jQuery.Event = function( src, props ) { + + // Allow instantiation without the 'new' keyword + if ( !( this instanceof jQuery.Event ) ) { + return new jQuery.Event( src, props ); + } + + // Event object + if ( src && src.type ) { + this.originalEvent = src; + this.type = src.type; + + // Events bubbling up the document may have been marked as prevented + // by a handler lower down the tree; reflect the correct value. + this.isDefaultPrevented = src.defaultPrevented || + src.defaultPrevented === undefined && + + // Support: Android <=2.3 only + src.returnValue === false ? + returnTrue : + returnFalse; + + // Create target properties + // Support: Safari <=6 - 7 only + // Target should not be a text node (#504, #13143) + this.target = ( src.target && src.target.nodeType === 3 ) ? + src.target.parentNode : + src.target; + + this.currentTarget = src.currentTarget; + this.relatedTarget = src.relatedTarget; + + // Event type + } else { + this.type = src; + } + + // Put explicitly provided properties onto the event object + if ( props ) { + jQuery.extend( this, props ); + } + + // Create a timestamp if incoming event doesn't have one + this.timeStamp = src && src.timeStamp || Date.now(); + + // Mark it as fixed + this[ jQuery.expando ] = true; +}; + +// jQuery.Event is based on DOM3 Events as specified by the ECMAScript Language Binding +// https://www.w3.org/TR/2003/WD-DOM-Level-3-Events-20030331/ecma-script-binding.html +jQuery.Event.prototype = { + constructor: jQuery.Event, + isDefaultPrevented: returnFalse, + isPropagationStopped: returnFalse, + isImmediatePropagationStopped: returnFalse, + isSimulated: false, + + preventDefault: function() { + var e = this.originalEvent; + + this.isDefaultPrevented = returnTrue; + + if ( e && !this.isSimulated ) { + e.preventDefault(); + } + }, + stopPropagation: function() { + var e = this.originalEvent; + + this.isPropagationStopped = returnTrue; + + if ( e && !this.isSimulated ) { + e.stopPropagation(); + } + }, + stopImmediatePropagation: function() { + var e = this.originalEvent; + + this.isImmediatePropagationStopped = returnTrue; + + if ( e && !this.isSimulated ) { + e.stopImmediatePropagation(); + } + + this.stopPropagation(); + } +}; + +// Includes all common event props including KeyEvent and MouseEvent specific props +jQuery.each( { + altKey: true, + bubbles: true, + cancelable: true, + changedTouches: true, + ctrlKey: true, + detail: true, + eventPhase: true, + metaKey: true, + pageX: true, + pageY: true, + shiftKey: true, + view: true, + "char": true, + code: true, + charCode: true, + key: true, + keyCode: true, + button: true, + buttons: true, + clientX: true, + clientY: true, + offsetX: true, + offsetY: true, + pointerId: true, + pointerType: true, + screenX: true, + screenY: true, + targetTouches: true, + toElement: true, + touches: true, + + which: function( event ) { + var button = event.button; + + // Add which for key events + if ( event.which == null && rkeyEvent.test( event.type ) ) { + return event.charCode != null ? event.charCode : event.keyCode; + } + + // Add which for click: 1 === left; 2 === middle; 3 === right + if ( !event.which && button !== undefined && rmouseEvent.test( event.type ) ) { + if ( button & 1 ) { + return 1; + } + + if ( button & 2 ) { + return 3; + } + + if ( button & 4 ) { + return 2; + } + + return 0; + } + + return event.which; + } +}, jQuery.event.addProp ); + +jQuery.each( { focus: "focusin", blur: "focusout" }, function( type, delegateType ) { + jQuery.event.special[ type ] = { + + // Utilize native event if possible so blur/focus sequence is correct + setup: function() { + + // Claim the first handler + // dataPriv.set( this, "focus", ... ) + // dataPriv.set( this, "blur", ... ) + leverageNative( this, type, expectSync ); + + // Return false to allow normal processing in the caller + return false; + }, + trigger: function() { + + // Force setup before trigger + leverageNative( this, type ); + + // Return non-false to allow normal event-path propagation + return true; + }, + + delegateType: delegateType + }; +} ); + +// Create mouseenter/leave events using mouseover/out and event-time checks +// so that event delegation works in jQuery. +// Do the same for pointerenter/pointerleave and pointerover/pointerout +// +// Support: Safari 7 only +// Safari sends mouseenter too often; see: +// https://bugs.chromium.org/p/chromium/issues/detail?id=470258 +// for the description of the bug (it existed in older Chrome versions as well). +jQuery.each( { + mouseenter: "mouseover", + mouseleave: "mouseout", + pointerenter: "pointerover", + pointerleave: "pointerout" +}, function( orig, fix ) { + jQuery.event.special[ orig ] = { + delegateType: fix, + bindType: fix, + + handle: function( event ) { + var ret, + target = this, + related = event.relatedTarget, + handleObj = event.handleObj; + + // For mouseenter/leave call the handler if related is outside the target. + // NB: No relatedTarget if the mouse left/entered the browser window + if ( !related || ( related !== target && !jQuery.contains( target, related ) ) ) { + event.type = handleObj.origType; + ret = handleObj.handler.apply( this, arguments ); + event.type = fix; + } + return ret; + } + }; +} ); + +jQuery.fn.extend( { + + on: function( types, selector, data, fn ) { + return on( this, types, selector, data, fn ); + }, + one: function( types, selector, data, fn ) { + return on( this, types, selector, data, fn, 1 ); + }, + off: function( types, selector, fn ) { + var handleObj, type; + if ( types && types.preventDefault && types.handleObj ) { + + // ( event ) dispatched jQuery.Event + handleObj = types.handleObj; + jQuery( types.delegateTarget ).off( + handleObj.namespace ? + handleObj.origType + "." + handleObj.namespace : + handleObj.origType, + handleObj.selector, + handleObj.handler + ); + return this; + } + if ( typeof types === "object" ) { + + // ( types-object [, selector] ) + for ( type in types ) { + this.off( type, selector, types[ type ] ); + } + return this; + } + if ( selector === false || typeof selector === "function" ) { + + // ( types [, fn] ) + fn = selector; + selector = undefined; + } + if ( fn === false ) { + fn = returnFalse; + } + return this.each( function() { + jQuery.event.remove( this, types, fn, selector ); + } ); + } +} ); + + +var + + // Support: IE <=10 - 11, Edge 12 - 13 only + // In IE/Edge using regex groups here causes severe slowdowns. + // See https://connect.microsoft.com/IE/feedback/details/1736512/ + rnoInnerhtml = /\s*$/g; + +// Prefer a tbody over its parent table for containing new rows +function manipulationTarget( elem, content ) { + if ( nodeName( elem, "table" ) && + nodeName( content.nodeType !== 11 ? content : content.firstChild, "tr" ) ) { + + return jQuery( elem ).children( "tbody" )[ 0 ] || elem; + } + + return elem; +} + +// Replace/restore the type attribute of script elements for safe DOM manipulation +function disableScript( elem ) { + elem.type = ( elem.getAttribute( "type" ) !== null ) + "/" + elem.type; + return elem; +} +function restoreScript( elem ) { + if ( ( elem.type || "" ).slice( 0, 5 ) === "true/" ) { + elem.type = elem.type.slice( 5 ); + } else { + elem.removeAttribute( "type" ); + } + + return elem; +} + +function cloneCopyEvent( src, dest ) { + var i, l, type, pdataOld, udataOld, udataCur, events; + + if ( dest.nodeType !== 1 ) { + return; + } + + // 1. Copy private data: events, handlers, etc. + if ( dataPriv.hasData( src ) ) { + pdataOld = dataPriv.get( src ); + events = pdataOld.events; + + if ( events ) { + dataPriv.remove( dest, "handle events" ); + + for ( type in events ) { + for ( i = 0, l = events[ type ].length; i < l; i++ ) { + jQuery.event.add( dest, type, events[ type ][ i ] ); + } + } + } + } + + // 2. Copy user data + if ( dataUser.hasData( src ) ) { + udataOld = dataUser.access( src ); + udataCur = jQuery.extend( {}, udataOld ); + + dataUser.set( dest, udataCur ); + } +} + +// Fix IE bugs, see support tests +function fixInput( src, dest ) { + var nodeName = dest.nodeName.toLowerCase(); + + // Fails to persist the checked state of a cloned checkbox or radio button. + if ( nodeName === "input" && rcheckableType.test( src.type ) ) { + dest.checked = src.checked; + + // Fails to return the selected option to the default selected state when cloning options + } else if ( nodeName === "input" || nodeName === "textarea" ) { + dest.defaultValue = src.defaultValue; + } +} + +function domManip( collection, args, callback, ignored ) { + + // Flatten any nested arrays + args = flat( args ); + + var fragment, first, scripts, hasScripts, node, doc, + i = 0, + l = collection.length, + iNoClone = l - 1, + value = args[ 0 ], + valueIsFunction = isFunction( value ); + + // We can't cloneNode fragments that contain checked, in WebKit + if ( valueIsFunction || + ( l > 1 && typeof value === "string" && + !support.checkClone && rchecked.test( value ) ) ) { + return collection.each( function( index ) { + var self = collection.eq( index ); + if ( valueIsFunction ) { + args[ 0 ] = value.call( this, index, self.html() ); + } + domManip( self, args, callback, ignored ); + } ); + } + + if ( l ) { + fragment = buildFragment( args, collection[ 0 ].ownerDocument, false, collection, ignored ); + first = fragment.firstChild; + + if ( fragment.childNodes.length === 1 ) { + fragment = first; + } + + // Require either new content or an interest in ignored elements to invoke the callback + if ( first || ignored ) { + scripts = jQuery.map( getAll( fragment, "script" ), disableScript ); + hasScripts = scripts.length; + + // Use the original fragment for the last item + // instead of the first because it can end up + // being emptied incorrectly in certain situations (#8070). + for ( ; i < l; i++ ) { + node = fragment; + + if ( i !== iNoClone ) { + node = jQuery.clone( node, true, true ); + + // Keep references to cloned scripts for later restoration + if ( hasScripts ) { + + // Support: Android <=4.0 only, PhantomJS 1 only + // push.apply(_, arraylike) throws on ancient WebKit + jQuery.merge( scripts, getAll( node, "script" ) ); + } + } + + callback.call( collection[ i ], node, i ); + } + + if ( hasScripts ) { + doc = scripts[ scripts.length - 1 ].ownerDocument; + + // Reenable scripts + jQuery.map( scripts, restoreScript ); + + // Evaluate executable scripts on first document insertion + for ( i = 0; i < hasScripts; i++ ) { + node = scripts[ i ]; + if ( rscriptType.test( node.type || "" ) && + !dataPriv.access( node, "globalEval" ) && + jQuery.contains( doc, node ) ) { + + if ( node.src && ( node.type || "" ).toLowerCase() !== "module" ) { + + // Optional AJAX dependency, but won't run scripts if not present + if ( jQuery._evalUrl && !node.noModule ) { + jQuery._evalUrl( node.src, { + nonce: node.nonce || node.getAttribute( "nonce" ) + }, doc ); + } + } else { + DOMEval( node.textContent.replace( rcleanScript, "" ), node, doc ); + } + } + } + } + } + } + + return collection; +} + +function remove( elem, selector, keepData ) { + var node, + nodes = selector ? jQuery.filter( selector, elem ) : elem, + i = 0; + + for ( ; ( node = nodes[ i ] ) != null; i++ ) { + if ( !keepData && node.nodeType === 1 ) { + jQuery.cleanData( getAll( node ) ); + } + + if ( node.parentNode ) { + if ( keepData && isAttached( node ) ) { + setGlobalEval( getAll( node, "script" ) ); + } + node.parentNode.removeChild( node ); + } + } + + return elem; +} + +jQuery.extend( { + htmlPrefilter: function( html ) { + return html; + }, + + clone: function( elem, dataAndEvents, deepDataAndEvents ) { + var i, l, srcElements, destElements, + clone = elem.cloneNode( true ), + inPage = isAttached( elem ); + + // Fix IE cloning issues + if ( !support.noCloneChecked && ( elem.nodeType === 1 || elem.nodeType === 11 ) && + !jQuery.isXMLDoc( elem ) ) { + + // We eschew Sizzle here for performance reasons: https://jsperf.com/getall-vs-sizzle/2 + destElements = getAll( clone ); + srcElements = getAll( elem ); + + for ( i = 0, l = srcElements.length; i < l; i++ ) { + fixInput( srcElements[ i ], destElements[ i ] ); + } + } + + // Copy the events from the original to the clone + if ( dataAndEvents ) { + if ( deepDataAndEvents ) { + srcElements = srcElements || getAll( elem ); + destElements = destElements || getAll( clone ); + + for ( i = 0, l = srcElements.length; i < l; i++ ) { + cloneCopyEvent( srcElements[ i ], destElements[ i ] ); + } + } else { + cloneCopyEvent( elem, clone ); + } + } + + // Preserve script evaluation history + destElements = getAll( clone, "script" ); + if ( destElements.length > 0 ) { + setGlobalEval( destElements, !inPage && getAll( elem, "script" ) ); + } + + // Return the cloned set + return clone; + }, + + cleanData: function( elems ) { + var data, elem, type, + special = jQuery.event.special, + i = 0; + + for ( ; ( elem = elems[ i ] ) !== undefined; i++ ) { + if ( acceptData( elem ) ) { + if ( ( data = elem[ dataPriv.expando ] ) ) { + if ( data.events ) { + for ( type in data.events ) { + if ( special[ type ] ) { + jQuery.event.remove( elem, type ); + + // This is a shortcut to avoid jQuery.event.remove's overhead + } else { + jQuery.removeEvent( elem, type, data.handle ); + } + } + } + + // Support: Chrome <=35 - 45+ + // Assign undefined instead of using delete, see Data#remove + elem[ dataPriv.expando ] = undefined; + } + if ( elem[ dataUser.expando ] ) { + + // Support: Chrome <=35 - 45+ + // Assign undefined instead of using delete, see Data#remove + elem[ dataUser.expando ] = undefined; + } + } + } + } +} ); + +jQuery.fn.extend( { + detach: function( selector ) { + return remove( this, selector, true ); + }, + + remove: function( selector ) { + return remove( this, selector ); + }, + + text: function( value ) { + return access( this, function( value ) { + return value === undefined ? + jQuery.text( this ) : + this.empty().each( function() { + if ( this.nodeType === 1 || this.nodeType === 11 || this.nodeType === 9 ) { + this.textContent = value; + } + } ); + }, null, value, arguments.length ); + }, + + append: function() { + return domManip( this, arguments, function( elem ) { + if ( this.nodeType === 1 || this.nodeType === 11 || this.nodeType === 9 ) { + var target = manipulationTarget( this, elem ); + target.appendChild( elem ); + } + } ); + }, + + prepend: function() { + return domManip( this, arguments, function( elem ) { + if ( this.nodeType === 1 || this.nodeType === 11 || this.nodeType === 9 ) { + var target = manipulationTarget( this, elem ); + target.insertBefore( elem, target.firstChild ); + } + } ); + }, + + before: function() { + return domManip( this, arguments, function( elem ) { + if ( this.parentNode ) { + this.parentNode.insertBefore( elem, this ); + } + } ); + }, + + after: function() { + return domManip( this, arguments, function( elem ) { + if ( this.parentNode ) { + this.parentNode.insertBefore( elem, this.nextSibling ); + } + } ); + }, + + empty: function() { + var elem, + i = 0; + + for ( ; ( elem = this[ i ] ) != null; i++ ) { + if ( elem.nodeType === 1 ) { + + // Prevent memory leaks + jQuery.cleanData( getAll( elem, false ) ); + + // Remove any remaining nodes + elem.textContent = ""; + } + } + + return this; + }, + + clone: function( dataAndEvents, deepDataAndEvents ) { + dataAndEvents = dataAndEvents == null ? false : dataAndEvents; + deepDataAndEvents = deepDataAndEvents == null ? dataAndEvents : deepDataAndEvents; + + return this.map( function() { + return jQuery.clone( this, dataAndEvents, deepDataAndEvents ); + } ); + }, + + html: function( value ) { + return access( this, function( value ) { + var elem = this[ 0 ] || {}, + i = 0, + l = this.length; + + if ( value === undefined && elem.nodeType === 1 ) { + return elem.innerHTML; + } + + // See if we can take a shortcut and just use innerHTML + if ( typeof value === "string" && !rnoInnerhtml.test( value ) && + !wrapMap[ ( rtagName.exec( value ) || [ "", "" ] )[ 1 ].toLowerCase() ] ) { + + value = jQuery.htmlPrefilter( value ); + + try { + for ( ; i < l; i++ ) { + elem = this[ i ] || {}; + + // Remove element nodes and prevent memory leaks + if ( elem.nodeType === 1 ) { + jQuery.cleanData( getAll( elem, false ) ); + elem.innerHTML = value; + } + } + + elem = 0; + + // If using innerHTML throws an exception, use the fallback method + } catch ( e ) {} + } + + if ( elem ) { + this.empty().append( value ); + } + }, null, value, arguments.length ); + }, + + replaceWith: function() { + var ignored = []; + + // Make the changes, replacing each non-ignored context element with the new content + return domManip( this, arguments, function( elem ) { + var parent = this.parentNode; + + if ( jQuery.inArray( this, ignored ) < 0 ) { + jQuery.cleanData( getAll( this ) ); + if ( parent ) { + parent.replaceChild( elem, this ); + } + } + + // Force callback invocation + }, ignored ); + } +} ); + +jQuery.each( { + appendTo: "append", + prependTo: "prepend", + insertBefore: "before", + insertAfter: "after", + replaceAll: "replaceWith" +}, function( name, original ) { + jQuery.fn[ name ] = function( selector ) { + var elems, + ret = [], + insert = jQuery( selector ), + last = insert.length - 1, + i = 0; + + for ( ; i <= last; i++ ) { + elems = i === last ? this : this.clone( true ); + jQuery( insert[ i ] )[ original ]( elems ); + + // Support: Android <=4.0 only, PhantomJS 1 only + // .get() because push.apply(_, arraylike) throws on ancient WebKit + push.apply( ret, elems.get() ); + } + + return this.pushStack( ret ); + }; +} ); +var rnumnonpx = new RegExp( "^(" + pnum + ")(?!px)[a-z%]+$", "i" ); + +var getStyles = function( elem ) { + + // Support: IE <=11 only, Firefox <=30 (#15098, #14150) + // IE throws on elements created in popups + // FF meanwhile throws on frame elements through "defaultView.getComputedStyle" + var view = elem.ownerDocument.defaultView; + + if ( !view || !view.opener ) { + view = window; + } + + return view.getComputedStyle( elem ); + }; + +var swap = function( elem, options, callback ) { + var ret, name, + old = {}; + + // Remember the old values, and insert the new ones + for ( name in options ) { + old[ name ] = elem.style[ name ]; + elem.style[ name ] = options[ name ]; + } + + ret = callback.call( elem ); + + // Revert the old values + for ( name in options ) { + elem.style[ name ] = old[ name ]; + } + + return ret; +}; + + +var rboxStyle = new RegExp( cssExpand.join( "|" ), "i" ); + + + +( function() { + + // Executing both pixelPosition & boxSizingReliable tests require only one layout + // so they're executed at the same time to save the second computation. + function computeStyleTests() { + + // This is a singleton, we need to execute it only once + if ( !div ) { + return; + } + + container.style.cssText = "position:absolute;left:-11111px;width:60px;" + + "margin-top:1px;padding:0;border:0"; + div.style.cssText = + "position:relative;display:block;box-sizing:border-box;overflow:scroll;" + + "margin:auto;border:1px;padding:1px;" + + "width:60%;top:1%"; + documentElement.appendChild( container ).appendChild( div ); + + var divStyle = window.getComputedStyle( div ); + pixelPositionVal = divStyle.top !== "1%"; + + // Support: Android 4.0 - 4.3 only, Firefox <=3 - 44 + reliableMarginLeftVal = roundPixelMeasures( divStyle.marginLeft ) === 12; + + // Support: Android 4.0 - 4.3 only, Safari <=9.1 - 10.1, iOS <=7.0 - 9.3 + // Some styles come back with percentage values, even though they shouldn't + div.style.right = "60%"; + pixelBoxStylesVal = roundPixelMeasures( divStyle.right ) === 36; + + // Support: IE 9 - 11 only + // Detect misreporting of content dimensions for box-sizing:border-box elements + boxSizingReliableVal = roundPixelMeasures( divStyle.width ) === 36; + + // Support: IE 9 only + // Detect overflow:scroll screwiness (gh-3699) + // Support: Chrome <=64 + // Don't get tricked when zoom affects offsetWidth (gh-4029) + div.style.position = "absolute"; + scrollboxSizeVal = roundPixelMeasures( div.offsetWidth / 3 ) === 12; + + documentElement.removeChild( container ); + + // Nullify the div so it wouldn't be stored in the memory and + // it will also be a sign that checks already performed + div = null; + } + + function roundPixelMeasures( measure ) { + return Math.round( parseFloat( measure ) ); + } + + var pixelPositionVal, boxSizingReliableVal, scrollboxSizeVal, pixelBoxStylesVal, + reliableTrDimensionsVal, reliableMarginLeftVal, + container = document.createElement( "div" ), + div = document.createElement( "div" ); + + // Finish early in limited (non-browser) environments + if ( !div.style ) { + return; + } + + // Support: IE <=9 - 11 only + // Style of cloned element affects source element cloned (#8908) + div.style.backgroundClip = "content-box"; + div.cloneNode( true ).style.backgroundClip = ""; + support.clearCloneStyle = div.style.backgroundClip === "content-box"; + + jQuery.extend( support, { + boxSizingReliable: function() { + computeStyleTests(); + return boxSizingReliableVal; + }, + pixelBoxStyles: function() { + computeStyleTests(); + return pixelBoxStylesVal; + }, + pixelPosition: function() { + computeStyleTests(); + return pixelPositionVal; + }, + reliableMarginLeft: function() { + computeStyleTests(); + return reliableMarginLeftVal; + }, + scrollboxSize: function() { + computeStyleTests(); + return scrollboxSizeVal; + }, + + // Support: IE 9 - 11+, Edge 15 - 18+ + // IE/Edge misreport `getComputedStyle` of table rows with width/height + // set in CSS while `offset*` properties report correct values. + // Behavior in IE 9 is more subtle than in newer versions & it passes + // some versions of this test; make sure not to make it pass there! + reliableTrDimensions: function() { + var table, tr, trChild, trStyle; + if ( reliableTrDimensionsVal == null ) { + table = document.createElement( "table" ); + tr = document.createElement( "tr" ); + trChild = document.createElement( "div" ); + + table.style.cssText = "position:absolute;left:-11111px"; + tr.style.height = "1px"; + trChild.style.height = "9px"; + + documentElement + .appendChild( table ) + .appendChild( tr ) + .appendChild( trChild ); + + trStyle = window.getComputedStyle( tr ); + reliableTrDimensionsVal = parseInt( trStyle.height ) > 3; + + documentElement.removeChild( table ); + } + return reliableTrDimensionsVal; + } + } ); +} )(); + + +function curCSS( elem, name, computed ) { + var width, minWidth, maxWidth, ret, + + // Support: Firefox 51+ + // Retrieving style before computed somehow + // fixes an issue with getting wrong values + // on detached elements + style = elem.style; + + computed = computed || getStyles( elem ); + + // getPropertyValue is needed for: + // .css('filter') (IE 9 only, #12537) + // .css('--customProperty) (#3144) + if ( computed ) { + ret = computed.getPropertyValue( name ) || computed[ name ]; + + if ( ret === "" && !isAttached( elem ) ) { + ret = jQuery.style( elem, name ); + } + + // A tribute to the "awesome hack by Dean Edwards" + // Android Browser returns percentage for some values, + // but width seems to be reliably pixels. + // This is against the CSSOM draft spec: + // https://drafts.csswg.org/cssom/#resolved-values + if ( !support.pixelBoxStyles() && rnumnonpx.test( ret ) && rboxStyle.test( name ) ) { + + // Remember the original values + width = style.width; + minWidth = style.minWidth; + maxWidth = style.maxWidth; + + // Put in the new values to get a computed value out + style.minWidth = style.maxWidth = style.width = ret; + ret = computed.width; + + // Revert the changed values + style.width = width; + style.minWidth = minWidth; + style.maxWidth = maxWidth; + } + } + + return ret !== undefined ? + + // Support: IE <=9 - 11 only + // IE returns zIndex value as an integer. + ret + "" : + ret; +} + + +function addGetHookIf( conditionFn, hookFn ) { + + // Define the hook, we'll check on the first run if it's really needed. + return { + get: function() { + if ( conditionFn() ) { + + // Hook not needed (or it's not possible to use it due + // to missing dependency), remove it. + delete this.get; + return; + } + + // Hook needed; redefine it so that the support test is not executed again. + return ( this.get = hookFn ).apply( this, arguments ); + } + }; +} + + +var cssPrefixes = [ "Webkit", "Moz", "ms" ], + emptyStyle = document.createElement( "div" ).style, + vendorProps = {}; + +// Return a vendor-prefixed property or undefined +function vendorPropName( name ) { + + // Check for vendor prefixed names + var capName = name[ 0 ].toUpperCase() + name.slice( 1 ), + i = cssPrefixes.length; + + while ( i-- ) { + name = cssPrefixes[ i ] + capName; + if ( name in emptyStyle ) { + return name; + } + } +} + +// Return a potentially-mapped jQuery.cssProps or vendor prefixed property +function finalPropName( name ) { + var final = jQuery.cssProps[ name ] || vendorProps[ name ]; + + if ( final ) { + return final; + } + if ( name in emptyStyle ) { + return name; + } + return vendorProps[ name ] = vendorPropName( name ) || name; +} + + +var + + // Swappable if display is none or starts with table + // except "table", "table-cell", or "table-caption" + // See here for display values: https://developer.mozilla.org/en-US/docs/CSS/display + rdisplayswap = /^(none|table(?!-c[ea]).+)/, + rcustomProp = /^--/, + cssShow = { position: "absolute", visibility: "hidden", display: "block" }, + cssNormalTransform = { + letterSpacing: "0", + fontWeight: "400" + }; + +function setPositiveNumber( _elem, value, subtract ) { + + // Any relative (+/-) values have already been + // normalized at this point + var matches = rcssNum.exec( value ); + return matches ? + + // Guard against undefined "subtract", e.g., when used as in cssHooks + Math.max( 0, matches[ 2 ] - ( subtract || 0 ) ) + ( matches[ 3 ] || "px" ) : + value; +} + +function boxModelAdjustment( elem, dimension, box, isBorderBox, styles, computedVal ) { + var i = dimension === "width" ? 1 : 0, + extra = 0, + delta = 0; + + // Adjustment may not be necessary + if ( box === ( isBorderBox ? "border" : "content" ) ) { + return 0; + } + + for ( ; i < 4; i += 2 ) { + + // Both box models exclude margin + if ( box === "margin" ) { + delta += jQuery.css( elem, box + cssExpand[ i ], true, styles ); + } + + // If we get here with a content-box, we're seeking "padding" or "border" or "margin" + if ( !isBorderBox ) { + + // Add padding + delta += jQuery.css( elem, "padding" + cssExpand[ i ], true, styles ); + + // For "border" or "margin", add border + if ( box !== "padding" ) { + delta += jQuery.css( elem, "border" + cssExpand[ i ] + "Width", true, styles ); + + // But still keep track of it otherwise + } else { + extra += jQuery.css( elem, "border" + cssExpand[ i ] + "Width", true, styles ); + } + + // If we get here with a border-box (content + padding + border), we're seeking "content" or + // "padding" or "margin" + } else { + + // For "content", subtract padding + if ( box === "content" ) { + delta -= jQuery.css( elem, "padding" + cssExpand[ i ], true, styles ); + } + + // For "content" or "padding", subtract border + if ( box !== "margin" ) { + delta -= jQuery.css( elem, "border" + cssExpand[ i ] + "Width", true, styles ); + } + } + } + + // Account for positive content-box scroll gutter when requested by providing computedVal + if ( !isBorderBox && computedVal >= 0 ) { + + // offsetWidth/offsetHeight is a rounded sum of content, padding, scroll gutter, and border + // Assuming integer scroll gutter, subtract the rest and round down + delta += Math.max( 0, Math.ceil( + elem[ "offset" + dimension[ 0 ].toUpperCase() + dimension.slice( 1 ) ] - + computedVal - + delta - + extra - + 0.5 + + // If offsetWidth/offsetHeight is unknown, then we can't determine content-box scroll gutter + // Use an explicit zero to avoid NaN (gh-3964) + ) ) || 0; + } + + return delta; +} + +function getWidthOrHeight( elem, dimension, extra ) { + + // Start with computed style + var styles = getStyles( elem ), + + // To avoid forcing a reflow, only fetch boxSizing if we need it (gh-4322). + // Fake content-box until we know it's needed to know the true value. + boxSizingNeeded = !support.boxSizingReliable() || extra, + isBorderBox = boxSizingNeeded && + jQuery.css( elem, "boxSizing", false, styles ) === "border-box", + valueIsBorderBox = isBorderBox, + + val = curCSS( elem, dimension, styles ), + offsetProp = "offset" + dimension[ 0 ].toUpperCase() + dimension.slice( 1 ); + + // Support: Firefox <=54 + // Return a confounding non-pixel value or feign ignorance, as appropriate. + if ( rnumnonpx.test( val ) ) { + if ( !extra ) { + return val; + } + val = "auto"; + } + + + // Support: IE 9 - 11 only + // Use offsetWidth/offsetHeight for when box sizing is unreliable. + // In those cases, the computed value can be trusted to be border-box. + if ( ( !support.boxSizingReliable() && isBorderBox || + + // Support: IE 10 - 11+, Edge 15 - 18+ + // IE/Edge misreport `getComputedStyle` of table rows with width/height + // set in CSS while `offset*` properties report correct values. + // Interestingly, in some cases IE 9 doesn't suffer from this issue. + !support.reliableTrDimensions() && nodeName( elem, "tr" ) || + + // Fall back to offsetWidth/offsetHeight when value is "auto" + // This happens for inline elements with no explicit setting (gh-3571) + val === "auto" || + + // Support: Android <=4.1 - 4.3 only + // Also use offsetWidth/offsetHeight for misreported inline dimensions (gh-3602) + !parseFloat( val ) && jQuery.css( elem, "display", false, styles ) === "inline" ) && + + // Make sure the element is visible & connected + elem.getClientRects().length ) { + + isBorderBox = jQuery.css( elem, "boxSizing", false, styles ) === "border-box"; + + // Where available, offsetWidth/offsetHeight approximate border box dimensions. + // Where not available (e.g., SVG), assume unreliable box-sizing and interpret the + // retrieved value as a content box dimension. + valueIsBorderBox = offsetProp in elem; + if ( valueIsBorderBox ) { + val = elem[ offsetProp ]; + } + } + + // Normalize "" and auto + val = parseFloat( val ) || 0; + + // Adjust for the element's box model + return ( val + + boxModelAdjustment( + elem, + dimension, + extra || ( isBorderBox ? "border" : "content" ), + valueIsBorderBox, + styles, + + // Provide the current computed size to request scroll gutter calculation (gh-3589) + val + ) + ) + "px"; +} + +jQuery.extend( { + + // Add in style property hooks for overriding the default + // behavior of getting and setting a style property + cssHooks: { + opacity: { + get: function( elem, computed ) { + if ( computed ) { + + // We should always get a number back from opacity + var ret = curCSS( elem, "opacity" ); + return ret === "" ? "1" : ret; + } + } + } + }, + + // Don't automatically add "px" to these possibly-unitless properties + cssNumber: { + "animationIterationCount": true, + "columnCount": true, + "fillOpacity": true, + "flexGrow": true, + "flexShrink": true, + "fontWeight": true, + "gridArea": true, + "gridColumn": true, + "gridColumnEnd": true, + "gridColumnStart": true, + "gridRow": true, + "gridRowEnd": true, + "gridRowStart": true, + "lineHeight": true, + "opacity": true, + "order": true, + "orphans": true, + "widows": true, + "zIndex": true, + "zoom": true + }, + + // Add in properties whose names you wish to fix before + // setting or getting the value + cssProps: {}, + + // Get and set the style property on a DOM Node + style: function( elem, name, value, extra ) { + + // Don't set styles on text and comment nodes + if ( !elem || elem.nodeType === 3 || elem.nodeType === 8 || !elem.style ) { + return; + } + + // Make sure that we're working with the right name + var ret, type, hooks, + origName = camelCase( name ), + isCustomProp = rcustomProp.test( name ), + style = elem.style; + + // Make sure that we're working with the right name. We don't + // want to query the value if it is a CSS custom property + // since they are user-defined. + if ( !isCustomProp ) { + name = finalPropName( origName ); + } + + // Gets hook for the prefixed version, then unprefixed version + hooks = jQuery.cssHooks[ name ] || jQuery.cssHooks[ origName ]; + + // Check if we're setting a value + if ( value !== undefined ) { + type = typeof value; + + // Convert "+=" or "-=" to relative numbers (#7345) + if ( type === "string" && ( ret = rcssNum.exec( value ) ) && ret[ 1 ] ) { + value = adjustCSS( elem, name, ret ); + + // Fixes bug #9237 + type = "number"; + } + + // Make sure that null and NaN values aren't set (#7116) + if ( value == null || value !== value ) { + return; + } + + // If a number was passed in, add the unit (except for certain CSS properties) + // The isCustomProp check can be removed in jQuery 4.0 when we only auto-append + // "px" to a few hardcoded values. + if ( type === "number" && !isCustomProp ) { + value += ret && ret[ 3 ] || ( jQuery.cssNumber[ origName ] ? "" : "px" ); + } + + // background-* props affect original clone's values + if ( !support.clearCloneStyle && value === "" && name.indexOf( "background" ) === 0 ) { + style[ name ] = "inherit"; + } + + // If a hook was provided, use that value, otherwise just set the specified value + if ( !hooks || !( "set" in hooks ) || + ( value = hooks.set( elem, value, extra ) ) !== undefined ) { + + if ( isCustomProp ) { + style.setProperty( name, value ); + } else { + style[ name ] = value; + } + } + + } else { + + // If a hook was provided get the non-computed value from there + if ( hooks && "get" in hooks && + ( ret = hooks.get( elem, false, extra ) ) !== undefined ) { + + return ret; + } + + // Otherwise just get the value from the style object + return style[ name ]; + } + }, + + css: function( elem, name, extra, styles ) { + var val, num, hooks, + origName = camelCase( name ), + isCustomProp = rcustomProp.test( name ); + + // Make sure that we're working with the right name. We don't + // want to modify the value if it is a CSS custom property + // since they are user-defined. + if ( !isCustomProp ) { + name = finalPropName( origName ); + } + + // Try prefixed name followed by the unprefixed name + hooks = jQuery.cssHooks[ name ] || jQuery.cssHooks[ origName ]; + + // If a hook was provided get the computed value from there + if ( hooks && "get" in hooks ) { + val = hooks.get( elem, true, extra ); + } + + // Otherwise, if a way to get the computed value exists, use that + if ( val === undefined ) { + val = curCSS( elem, name, styles ); + } + + // Convert "normal" to computed value + if ( val === "normal" && name in cssNormalTransform ) { + val = cssNormalTransform[ name ]; + } + + // Make numeric if forced or a qualifier was provided and val looks numeric + if ( extra === "" || extra ) { + num = parseFloat( val ); + return extra === true || isFinite( num ) ? num || 0 : val; + } + + return val; + } +} ); + +jQuery.each( [ "height", "width" ], function( _i, dimension ) { + jQuery.cssHooks[ dimension ] = { + get: function( elem, computed, extra ) { + if ( computed ) { + + // Certain elements can have dimension info if we invisibly show them + // but it must have a current display style that would benefit + return rdisplayswap.test( jQuery.css( elem, "display" ) ) && + + // Support: Safari 8+ + // Table columns in Safari have non-zero offsetWidth & zero + // getBoundingClientRect().width unless display is changed. + // Support: IE <=11 only + // Running getBoundingClientRect on a disconnected node + // in IE throws an error. + ( !elem.getClientRects().length || !elem.getBoundingClientRect().width ) ? + swap( elem, cssShow, function() { + return getWidthOrHeight( elem, dimension, extra ); + } ) : + getWidthOrHeight( elem, dimension, extra ); + } + }, + + set: function( elem, value, extra ) { + var matches, + styles = getStyles( elem ), + + // Only read styles.position if the test has a chance to fail + // to avoid forcing a reflow. + scrollboxSizeBuggy = !support.scrollboxSize() && + styles.position === "absolute", + + // To avoid forcing a reflow, only fetch boxSizing if we need it (gh-3991) + boxSizingNeeded = scrollboxSizeBuggy || extra, + isBorderBox = boxSizingNeeded && + jQuery.css( elem, "boxSizing", false, styles ) === "border-box", + subtract = extra ? + boxModelAdjustment( + elem, + dimension, + extra, + isBorderBox, + styles + ) : + 0; + + // Account for unreliable border-box dimensions by comparing offset* to computed and + // faking a content-box to get border and padding (gh-3699) + if ( isBorderBox && scrollboxSizeBuggy ) { + subtract -= Math.ceil( + elem[ "offset" + dimension[ 0 ].toUpperCase() + dimension.slice( 1 ) ] - + parseFloat( styles[ dimension ] ) - + boxModelAdjustment( elem, dimension, "border", false, styles ) - + 0.5 + ); + } + + // Convert to pixels if value adjustment is needed + if ( subtract && ( matches = rcssNum.exec( value ) ) && + ( matches[ 3 ] || "px" ) !== "px" ) { + + elem.style[ dimension ] = value; + value = jQuery.css( elem, dimension ); + } + + return setPositiveNumber( elem, value, subtract ); + } + }; +} ); + +jQuery.cssHooks.marginLeft = addGetHookIf( support.reliableMarginLeft, + function( elem, computed ) { + if ( computed ) { + return ( parseFloat( curCSS( elem, "marginLeft" ) ) || + elem.getBoundingClientRect().left - + swap( elem, { marginLeft: 0 }, function() { + return elem.getBoundingClientRect().left; + } ) + ) + "px"; + } + } +); + +// These hooks are used by animate to expand properties +jQuery.each( { + margin: "", + padding: "", + border: "Width" +}, function( prefix, suffix ) { + jQuery.cssHooks[ prefix + suffix ] = { + expand: function( value ) { + var i = 0, + expanded = {}, + + // Assumes a single number if not a string + parts = typeof value === "string" ? value.split( " " ) : [ value ]; + + for ( ; i < 4; i++ ) { + expanded[ prefix + cssExpand[ i ] + suffix ] = + parts[ i ] || parts[ i - 2 ] || parts[ 0 ]; + } + + return expanded; + } + }; + + if ( prefix !== "margin" ) { + jQuery.cssHooks[ prefix + suffix ].set = setPositiveNumber; + } +} ); + +jQuery.fn.extend( { + css: function( name, value ) { + return access( this, function( elem, name, value ) { + var styles, len, + map = {}, + i = 0; + + if ( Array.isArray( name ) ) { + styles = getStyles( elem ); + len = name.length; + + for ( ; i < len; i++ ) { + map[ name[ i ] ] = jQuery.css( elem, name[ i ], false, styles ); + } + + return map; + } + + return value !== undefined ? + jQuery.style( elem, name, value ) : + jQuery.css( elem, name ); + }, name, value, arguments.length > 1 ); + } +} ); + + +function Tween( elem, options, prop, end, easing ) { + return new Tween.prototype.init( elem, options, prop, end, easing ); +} +jQuery.Tween = Tween; + +Tween.prototype = { + constructor: Tween, + init: function( elem, options, prop, end, easing, unit ) { + this.elem = elem; + this.prop = prop; + this.easing = easing || jQuery.easing._default; + this.options = options; + this.start = this.now = this.cur(); + this.end = end; + this.unit = unit || ( jQuery.cssNumber[ prop ] ? "" : "px" ); + }, + cur: function() { + var hooks = Tween.propHooks[ this.prop ]; + + return hooks && hooks.get ? + hooks.get( this ) : + Tween.propHooks._default.get( this ); + }, + run: function( percent ) { + var eased, + hooks = Tween.propHooks[ this.prop ]; + + if ( this.options.duration ) { + this.pos = eased = jQuery.easing[ this.easing ]( + percent, this.options.duration * percent, 0, 1, this.options.duration + ); + } else { + this.pos = eased = percent; + } + this.now = ( this.end - this.start ) * eased + this.start; + + if ( this.options.step ) { + this.options.step.call( this.elem, this.now, this ); + } + + if ( hooks && hooks.set ) { + hooks.set( this ); + } else { + Tween.propHooks._default.set( this ); + } + return this; + } +}; + +Tween.prototype.init.prototype = Tween.prototype; + +Tween.propHooks = { + _default: { + get: function( tween ) { + var result; + + // Use a property on the element directly when it is not a DOM element, + // or when there is no matching style property that exists. + if ( tween.elem.nodeType !== 1 || + tween.elem[ tween.prop ] != null && tween.elem.style[ tween.prop ] == null ) { + return tween.elem[ tween.prop ]; + } + + // Passing an empty string as a 3rd parameter to .css will automatically + // attempt a parseFloat and fallback to a string if the parse fails. + // Simple values such as "10px" are parsed to Float; + // complex values such as "rotate(1rad)" are returned as-is. + result = jQuery.css( tween.elem, tween.prop, "" ); + + // Empty strings, null, undefined and "auto" are converted to 0. + return !result || result === "auto" ? 0 : result; + }, + set: function( tween ) { + + // Use step hook for back compat. + // Use cssHook if its there. + // Use .style if available and use plain properties where available. + if ( jQuery.fx.step[ tween.prop ] ) { + jQuery.fx.step[ tween.prop ]( tween ); + } else if ( tween.elem.nodeType === 1 && ( + jQuery.cssHooks[ tween.prop ] || + tween.elem.style[ finalPropName( tween.prop ) ] != null ) ) { + jQuery.style( tween.elem, tween.prop, tween.now + tween.unit ); + } else { + tween.elem[ tween.prop ] = tween.now; + } + } + } +}; + +// Support: IE <=9 only +// Panic based approach to setting things on disconnected nodes +Tween.propHooks.scrollTop = Tween.propHooks.scrollLeft = { + set: function( tween ) { + if ( tween.elem.nodeType && tween.elem.parentNode ) { + tween.elem[ tween.prop ] = tween.now; + } + } +}; + +jQuery.easing = { + linear: function( p ) { + return p; + }, + swing: function( p ) { + return 0.5 - Math.cos( p * Math.PI ) / 2; + }, + _default: "swing" +}; + +jQuery.fx = Tween.prototype.init; + +// Back compat <1.8 extension point +jQuery.fx.step = {}; + + + + +var + fxNow, inProgress, + rfxtypes = /^(?:toggle|show|hide)$/, + rrun = /queueHooks$/; + +function schedule() { + if ( inProgress ) { + if ( document.hidden === false && window.requestAnimationFrame ) { + window.requestAnimationFrame( schedule ); + } else { + window.setTimeout( schedule, jQuery.fx.interval ); + } + + jQuery.fx.tick(); + } +} + +// Animations created synchronously will run synchronously +function createFxNow() { + window.setTimeout( function() { + fxNow = undefined; + } ); + return ( fxNow = Date.now() ); +} + +// Generate parameters to create a standard animation +function genFx( type, includeWidth ) { + var which, + i = 0, + attrs = { height: type }; + + // If we include width, step value is 1 to do all cssExpand values, + // otherwise step value is 2 to skip over Left and Right + includeWidth = includeWidth ? 1 : 0; + for ( ; i < 4; i += 2 - includeWidth ) { + which = cssExpand[ i ]; + attrs[ "margin" + which ] = attrs[ "padding" + which ] = type; + } + + if ( includeWidth ) { + attrs.opacity = attrs.width = type; + } + + return attrs; +} + +function createTween( value, prop, animation ) { + var tween, + collection = ( Animation.tweeners[ prop ] || [] ).concat( Animation.tweeners[ "*" ] ), + index = 0, + length = collection.length; + for ( ; index < length; index++ ) { + if ( ( tween = collection[ index ].call( animation, prop, value ) ) ) { + + // We're done with this property + return tween; + } + } +} + +function defaultPrefilter( elem, props, opts ) { + var prop, value, toggle, hooks, oldfire, propTween, restoreDisplay, display, + isBox = "width" in props || "height" in props, + anim = this, + orig = {}, + style = elem.style, + hidden = elem.nodeType && isHiddenWithinTree( elem ), + dataShow = dataPriv.get( elem, "fxshow" ); + + // Queue-skipping animations hijack the fx hooks + if ( !opts.queue ) { + hooks = jQuery._queueHooks( elem, "fx" ); + if ( hooks.unqueued == null ) { + hooks.unqueued = 0; + oldfire = hooks.empty.fire; + hooks.empty.fire = function() { + if ( !hooks.unqueued ) { + oldfire(); + } + }; + } + hooks.unqueued++; + + anim.always( function() { + + // Ensure the complete handler is called before this completes + anim.always( function() { + hooks.unqueued--; + if ( !jQuery.queue( elem, "fx" ).length ) { + hooks.empty.fire(); + } + } ); + } ); + } + + // Detect show/hide animations + for ( prop in props ) { + value = props[ prop ]; + if ( rfxtypes.test( value ) ) { + delete props[ prop ]; + toggle = toggle || value === "toggle"; + if ( value === ( hidden ? "hide" : "show" ) ) { + + // Pretend to be hidden if this is a "show" and + // there is still data from a stopped show/hide + if ( value === "show" && dataShow && dataShow[ prop ] !== undefined ) { + hidden = true; + + // Ignore all other no-op show/hide data + } else { + continue; + } + } + orig[ prop ] = dataShow && dataShow[ prop ] || jQuery.style( elem, prop ); + } + } + + // Bail out if this is a no-op like .hide().hide() + propTween = !jQuery.isEmptyObject( props ); + if ( !propTween && jQuery.isEmptyObject( orig ) ) { + return; + } + + // Restrict "overflow" and "display" styles during box animations + if ( isBox && elem.nodeType === 1 ) { + + // Support: IE <=9 - 11, Edge 12 - 15 + // Record all 3 overflow attributes because IE does not infer the shorthand + // from identically-valued overflowX and overflowY and Edge just mirrors + // the overflowX value there. + opts.overflow = [ style.overflow, style.overflowX, style.overflowY ]; + + // Identify a display type, preferring old show/hide data over the CSS cascade + restoreDisplay = dataShow && dataShow.display; + if ( restoreDisplay == null ) { + restoreDisplay = dataPriv.get( elem, "display" ); + } + display = jQuery.css( elem, "display" ); + if ( display === "none" ) { + if ( restoreDisplay ) { + display = restoreDisplay; + } else { + + // Get nonempty value(s) by temporarily forcing visibility + showHide( [ elem ], true ); + restoreDisplay = elem.style.display || restoreDisplay; + display = jQuery.css( elem, "display" ); + showHide( [ elem ] ); + } + } + + // Animate inline elements as inline-block + if ( display === "inline" || display === "inline-block" && restoreDisplay != null ) { + if ( jQuery.css( elem, "float" ) === "none" ) { + + // Restore the original display value at the end of pure show/hide animations + if ( !propTween ) { + anim.done( function() { + style.display = restoreDisplay; + } ); + if ( restoreDisplay == null ) { + display = style.display; + restoreDisplay = display === "none" ? "" : display; + } + } + style.display = "inline-block"; + } + } + } + + if ( opts.overflow ) { + style.overflow = "hidden"; + anim.always( function() { + style.overflow = opts.overflow[ 0 ]; + style.overflowX = opts.overflow[ 1 ]; + style.overflowY = opts.overflow[ 2 ]; + } ); + } + + // Implement show/hide animations + propTween = false; + for ( prop in orig ) { + + // General show/hide setup for this element animation + if ( !propTween ) { + if ( dataShow ) { + if ( "hidden" in dataShow ) { + hidden = dataShow.hidden; + } + } else { + dataShow = dataPriv.access( elem, "fxshow", { display: restoreDisplay } ); + } + + // Store hidden/visible for toggle so `.stop().toggle()` "reverses" + if ( toggle ) { + dataShow.hidden = !hidden; + } + + // Show elements before animating them + if ( hidden ) { + showHide( [ elem ], true ); + } + + /* eslint-disable no-loop-func */ + + anim.done( function() { + + /* eslint-enable no-loop-func */ + + // The final step of a "hide" animation is actually hiding the element + if ( !hidden ) { + showHide( [ elem ] ); + } + dataPriv.remove( elem, "fxshow" ); + for ( prop in orig ) { + jQuery.style( elem, prop, orig[ prop ] ); + } + } ); + } + + // Per-property setup + propTween = createTween( hidden ? dataShow[ prop ] : 0, prop, anim ); + if ( !( prop in dataShow ) ) { + dataShow[ prop ] = propTween.start; + if ( hidden ) { + propTween.end = propTween.start; + propTween.start = 0; + } + } + } +} + +function propFilter( props, specialEasing ) { + var index, name, easing, value, hooks; + + // camelCase, specialEasing and expand cssHook pass + for ( index in props ) { + name = camelCase( index ); + easing = specialEasing[ name ]; + value = props[ index ]; + if ( Array.isArray( value ) ) { + easing = value[ 1 ]; + value = props[ index ] = value[ 0 ]; + } + + if ( index !== name ) { + props[ name ] = value; + delete props[ index ]; + } + + hooks = jQuery.cssHooks[ name ]; + if ( hooks && "expand" in hooks ) { + value = hooks.expand( value ); + delete props[ name ]; + + // Not quite $.extend, this won't overwrite existing keys. + // Reusing 'index' because we have the correct "name" + for ( index in value ) { + if ( !( index in props ) ) { + props[ index ] = value[ index ]; + specialEasing[ index ] = easing; + } + } + } else { + specialEasing[ name ] = easing; + } + } +} + +function Animation( elem, properties, options ) { + var result, + stopped, + index = 0, + length = Animation.prefilters.length, + deferred = jQuery.Deferred().always( function() { + + // Don't match elem in the :animated selector + delete tick.elem; + } ), + tick = function() { + if ( stopped ) { + return false; + } + var currentTime = fxNow || createFxNow(), + remaining = Math.max( 0, animation.startTime + animation.duration - currentTime ), + + // Support: Android 2.3 only + // Archaic crash bug won't allow us to use `1 - ( 0.5 || 0 )` (#12497) + temp = remaining / animation.duration || 0, + percent = 1 - temp, + index = 0, + length = animation.tweens.length; + + for ( ; index < length; index++ ) { + animation.tweens[ index ].run( percent ); + } + + deferred.notifyWith( elem, [ animation, percent, remaining ] ); + + // If there's more to do, yield + if ( percent < 1 && length ) { + return remaining; + } + + // If this was an empty animation, synthesize a final progress notification + if ( !length ) { + deferred.notifyWith( elem, [ animation, 1, 0 ] ); + } + + // Resolve the animation and report its conclusion + deferred.resolveWith( elem, [ animation ] ); + return false; + }, + animation = deferred.promise( { + elem: elem, + props: jQuery.extend( {}, properties ), + opts: jQuery.extend( true, { + specialEasing: {}, + easing: jQuery.easing._default + }, options ), + originalProperties: properties, + originalOptions: options, + startTime: fxNow || createFxNow(), + duration: options.duration, + tweens: [], + createTween: function( prop, end ) { + var tween = jQuery.Tween( elem, animation.opts, prop, end, + animation.opts.specialEasing[ prop ] || animation.opts.easing ); + animation.tweens.push( tween ); + return tween; + }, + stop: function( gotoEnd ) { + var index = 0, + + // If we are going to the end, we want to run all the tweens + // otherwise we skip this part + length = gotoEnd ? animation.tweens.length : 0; + if ( stopped ) { + return this; + } + stopped = true; + for ( ; index < length; index++ ) { + animation.tweens[ index ].run( 1 ); + } + + // Resolve when we played the last frame; otherwise, reject + if ( gotoEnd ) { + deferred.notifyWith( elem, [ animation, 1, 0 ] ); + deferred.resolveWith( elem, [ animation, gotoEnd ] ); + } else { + deferred.rejectWith( elem, [ animation, gotoEnd ] ); + } + return this; + } + } ), + props = animation.props; + + propFilter( props, animation.opts.specialEasing ); + + for ( ; index < length; index++ ) { + result = Animation.prefilters[ index ].call( animation, elem, props, animation.opts ); + if ( result ) { + if ( isFunction( result.stop ) ) { + jQuery._queueHooks( animation.elem, animation.opts.queue ).stop = + result.stop.bind( result ); + } + return result; + } + } + + jQuery.map( props, createTween, animation ); + + if ( isFunction( animation.opts.start ) ) { + animation.opts.start.call( elem, animation ); + } + + // Attach callbacks from options + animation + .progress( animation.opts.progress ) + .done( animation.opts.done, animation.opts.complete ) + .fail( animation.opts.fail ) + .always( animation.opts.always ); + + jQuery.fx.timer( + jQuery.extend( tick, { + elem: elem, + anim: animation, + queue: animation.opts.queue + } ) + ); + + return animation; +} + +jQuery.Animation = jQuery.extend( Animation, { + + tweeners: { + "*": [ function( prop, value ) { + var tween = this.createTween( prop, value ); + adjustCSS( tween.elem, prop, rcssNum.exec( value ), tween ); + return tween; + } ] + }, + + tweener: function( props, callback ) { + if ( isFunction( props ) ) { + callback = props; + props = [ "*" ]; + } else { + props = props.match( rnothtmlwhite ); + } + + var prop, + index = 0, + length = props.length; + + for ( ; index < length; index++ ) { + prop = props[ index ]; + Animation.tweeners[ prop ] = Animation.tweeners[ prop ] || []; + Animation.tweeners[ prop ].unshift( callback ); + } + }, + + prefilters: [ defaultPrefilter ], + + prefilter: function( callback, prepend ) { + if ( prepend ) { + Animation.prefilters.unshift( callback ); + } else { + Animation.prefilters.push( callback ); + } + } +} ); + +jQuery.speed = function( speed, easing, fn ) { + var opt = speed && typeof speed === "object" ? jQuery.extend( {}, speed ) : { + complete: fn || !fn && easing || + isFunction( speed ) && speed, + duration: speed, + easing: fn && easing || easing && !isFunction( easing ) && easing + }; + + // Go to the end state if fx are off + if ( jQuery.fx.off ) { + opt.duration = 0; + + } else { + if ( typeof opt.duration !== "number" ) { + if ( opt.duration in jQuery.fx.speeds ) { + opt.duration = jQuery.fx.speeds[ opt.duration ]; + + } else { + opt.duration = jQuery.fx.speeds._default; + } + } + } + + // Normalize opt.queue - true/undefined/null -> "fx" + if ( opt.queue == null || opt.queue === true ) { + opt.queue = "fx"; + } + + // Queueing + opt.old = opt.complete; + + opt.complete = function() { + if ( isFunction( opt.old ) ) { + opt.old.call( this ); + } + + if ( opt.queue ) { + jQuery.dequeue( this, opt.queue ); + } + }; + + return opt; +}; + +jQuery.fn.extend( { + fadeTo: function( speed, to, easing, callback ) { + + // Show any hidden elements after setting opacity to 0 + return this.filter( isHiddenWithinTree ).css( "opacity", 0 ).show() + + // Animate to the value specified + .end().animate( { opacity: to }, speed, easing, callback ); + }, + animate: function( prop, speed, easing, callback ) { + var empty = jQuery.isEmptyObject( prop ), + optall = jQuery.speed( speed, easing, callback ), + doAnimation = function() { + + // Operate on a copy of prop so per-property easing won't be lost + var anim = Animation( this, jQuery.extend( {}, prop ), optall ); + + // Empty animations, or finishing resolves immediately + if ( empty || dataPriv.get( this, "finish" ) ) { + anim.stop( true ); + } + }; + doAnimation.finish = doAnimation; + + return empty || optall.queue === false ? + this.each( doAnimation ) : + this.queue( optall.queue, doAnimation ); + }, + stop: function( type, clearQueue, gotoEnd ) { + var stopQueue = function( hooks ) { + var stop = hooks.stop; + delete hooks.stop; + stop( gotoEnd ); + }; + + if ( typeof type !== "string" ) { + gotoEnd = clearQueue; + clearQueue = type; + type = undefined; + } + if ( clearQueue ) { + this.queue( type || "fx", [] ); + } + + return this.each( function() { + var dequeue = true, + index = type != null && type + "queueHooks", + timers = jQuery.timers, + data = dataPriv.get( this ); + + if ( index ) { + if ( data[ index ] && data[ index ].stop ) { + stopQueue( data[ index ] ); + } + } else { + for ( index in data ) { + if ( data[ index ] && data[ index ].stop && rrun.test( index ) ) { + stopQueue( data[ index ] ); + } + } + } + + for ( index = timers.length; index--; ) { + if ( timers[ index ].elem === this && + ( type == null || timers[ index ].queue === type ) ) { + + timers[ index ].anim.stop( gotoEnd ); + dequeue = false; + timers.splice( index, 1 ); + } + } + + // Start the next in the queue if the last step wasn't forced. + // Timers currently will call their complete callbacks, which + // will dequeue but only if they were gotoEnd. + if ( dequeue || !gotoEnd ) { + jQuery.dequeue( this, type ); + } + } ); + }, + finish: function( type ) { + if ( type !== false ) { + type = type || "fx"; + } + return this.each( function() { + var index, + data = dataPriv.get( this ), + queue = data[ type + "queue" ], + hooks = data[ type + "queueHooks" ], + timers = jQuery.timers, + length = queue ? queue.length : 0; + + // Enable finishing flag on private data + data.finish = true; + + // Empty the queue first + jQuery.queue( this, type, [] ); + + if ( hooks && hooks.stop ) { + hooks.stop.call( this, true ); + } + + // Look for any active animations, and finish them + for ( index = timers.length; index--; ) { + if ( timers[ index ].elem === this && timers[ index ].queue === type ) { + timers[ index ].anim.stop( true ); + timers.splice( index, 1 ); + } + } + + // Look for any animations in the old queue and finish them + for ( index = 0; index < length; index++ ) { + if ( queue[ index ] && queue[ index ].finish ) { + queue[ index ].finish.call( this ); + } + } + + // Turn off finishing flag + delete data.finish; + } ); + } +} ); + +jQuery.each( [ "toggle", "show", "hide" ], function( _i, name ) { + var cssFn = jQuery.fn[ name ]; + jQuery.fn[ name ] = function( speed, easing, callback ) { + return speed == null || typeof speed === "boolean" ? + cssFn.apply( this, arguments ) : + this.animate( genFx( name, true ), speed, easing, callback ); + }; +} ); + +// Generate shortcuts for custom animations +jQuery.each( { + slideDown: genFx( "show" ), + slideUp: genFx( "hide" ), + slideToggle: genFx( "toggle" ), + fadeIn: { opacity: "show" }, + fadeOut: { opacity: "hide" }, + fadeToggle: { opacity: "toggle" } +}, function( name, props ) { + jQuery.fn[ name ] = function( speed, easing, callback ) { + return this.animate( props, speed, easing, callback ); + }; +} ); + +jQuery.timers = []; +jQuery.fx.tick = function() { + var timer, + i = 0, + timers = jQuery.timers; + + fxNow = Date.now(); + + for ( ; i < timers.length; i++ ) { + timer = timers[ i ]; + + // Run the timer and safely remove it when done (allowing for external removal) + if ( !timer() && timers[ i ] === timer ) { + timers.splice( i--, 1 ); + } + } + + if ( !timers.length ) { + jQuery.fx.stop(); + } + fxNow = undefined; +}; + +jQuery.fx.timer = function( timer ) { + jQuery.timers.push( timer ); + jQuery.fx.start(); +}; + +jQuery.fx.interval = 13; +jQuery.fx.start = function() { + if ( inProgress ) { + return; + } + + inProgress = true; + schedule(); +}; + +jQuery.fx.stop = function() { + inProgress = null; +}; + +jQuery.fx.speeds = { + slow: 600, + fast: 200, + + // Default speed + _default: 400 +}; + + +// Based off of the plugin by Clint Helfers, with permission. +// https://web.archive.org/web/20100324014747/http://blindsignals.com/index.php/2009/07/jquery-delay/ +jQuery.fn.delay = function( time, type ) { + time = jQuery.fx ? jQuery.fx.speeds[ time ] || time : time; + type = type || "fx"; + + return this.queue( type, function( next, hooks ) { + var timeout = window.setTimeout( next, time ); + hooks.stop = function() { + window.clearTimeout( timeout ); + }; + } ); +}; + + +( function() { + var input = document.createElement( "input" ), + select = document.createElement( "select" ), + opt = select.appendChild( document.createElement( "option" ) ); + + input.type = "checkbox"; + + // Support: Android <=4.3 only + // Default value for a checkbox should be "on" + support.checkOn = input.value !== ""; + + // Support: IE <=11 only + // Must access selectedIndex to make default options select + support.optSelected = opt.selected; + + // Support: IE <=11 only + // An input loses its value after becoming a radio + input = document.createElement( "input" ); + input.value = "t"; + input.type = "radio"; + support.radioValue = input.value === "t"; +} )(); + + +var boolHook, + attrHandle = jQuery.expr.attrHandle; + +jQuery.fn.extend( { + attr: function( name, value ) { + return access( this, jQuery.attr, name, value, arguments.length > 1 ); + }, + + removeAttr: function( name ) { + return this.each( function() { + jQuery.removeAttr( this, name ); + } ); + } +} ); + +jQuery.extend( { + attr: function( elem, name, value ) { + var ret, hooks, + nType = elem.nodeType; + + // Don't get/set attributes on text, comment and attribute nodes + if ( nType === 3 || nType === 8 || nType === 2 ) { + return; + } + + // Fallback to prop when attributes are not supported + if ( typeof elem.getAttribute === "undefined" ) { + return jQuery.prop( elem, name, value ); + } + + // Attribute hooks are determined by the lowercase version + // Grab necessary hook if one is defined + if ( nType !== 1 || !jQuery.isXMLDoc( elem ) ) { + hooks = jQuery.attrHooks[ name.toLowerCase() ] || + ( jQuery.expr.match.bool.test( name ) ? boolHook : undefined ); + } + + if ( value !== undefined ) { + if ( value === null ) { + jQuery.removeAttr( elem, name ); + return; + } + + if ( hooks && "set" in hooks && + ( ret = hooks.set( elem, value, name ) ) !== undefined ) { + return ret; + } + + elem.setAttribute( name, value + "" ); + return value; + } + + if ( hooks && "get" in hooks && ( ret = hooks.get( elem, name ) ) !== null ) { + return ret; + } + + ret = jQuery.find.attr( elem, name ); + + // Non-existent attributes return null, we normalize to undefined + return ret == null ? undefined : ret; + }, + + attrHooks: { + type: { + set: function( elem, value ) { + if ( !support.radioValue && value === "radio" && + nodeName( elem, "input" ) ) { + var val = elem.value; + elem.setAttribute( "type", value ); + if ( val ) { + elem.value = val; + } + return value; + } + } + } + }, + + removeAttr: function( elem, value ) { + var name, + i = 0, + + // Attribute names can contain non-HTML whitespace characters + // https://html.spec.whatwg.org/multipage/syntax.html#attributes-2 + attrNames = value && value.match( rnothtmlwhite ); + + if ( attrNames && elem.nodeType === 1 ) { + while ( ( name = attrNames[ i++ ] ) ) { + elem.removeAttribute( name ); + } + } + } +} ); + +// Hooks for boolean attributes +boolHook = { + set: function( elem, value, name ) { + if ( value === false ) { + + // Remove boolean attributes when set to false + jQuery.removeAttr( elem, name ); + } else { + elem.setAttribute( name, name ); + } + return name; + } +}; + +jQuery.each( jQuery.expr.match.bool.source.match( /\w+/g ), function( _i, name ) { + var getter = attrHandle[ name ] || jQuery.find.attr; + + attrHandle[ name ] = function( elem, name, isXML ) { + var ret, handle, + lowercaseName = name.toLowerCase(); + + if ( !isXML ) { + + // Avoid an infinite loop by temporarily removing this function from the getter + handle = attrHandle[ lowercaseName ]; + attrHandle[ lowercaseName ] = ret; + ret = getter( elem, name, isXML ) != null ? + lowercaseName : + null; + attrHandle[ lowercaseName ] = handle; + } + return ret; + }; +} ); + + + + +var rfocusable = /^(?:input|select|textarea|button)$/i, + rclickable = /^(?:a|area)$/i; + +jQuery.fn.extend( { + prop: function( name, value ) { + return access( this, jQuery.prop, name, value, arguments.length > 1 ); + }, + + removeProp: function( name ) { + return this.each( function() { + delete this[ jQuery.propFix[ name ] || name ]; + } ); + } +} ); + +jQuery.extend( { + prop: function( elem, name, value ) { + var ret, hooks, + nType = elem.nodeType; + + // Don't get/set properties on text, comment and attribute nodes + if ( nType === 3 || nType === 8 || nType === 2 ) { + return; + } + + if ( nType !== 1 || !jQuery.isXMLDoc( elem ) ) { + + // Fix name and attach hooks + name = jQuery.propFix[ name ] || name; + hooks = jQuery.propHooks[ name ]; + } + + if ( value !== undefined ) { + if ( hooks && "set" in hooks && + ( ret = hooks.set( elem, value, name ) ) !== undefined ) { + return ret; + } + + return ( elem[ name ] = value ); + } + + if ( hooks && "get" in hooks && ( ret = hooks.get( elem, name ) ) !== null ) { + return ret; + } + + return elem[ name ]; + }, + + propHooks: { + tabIndex: { + get: function( elem ) { + + // Support: IE <=9 - 11 only + // elem.tabIndex doesn't always return the + // correct value when it hasn't been explicitly set + // https://web.archive.org/web/20141116233347/http://fluidproject.org/blog/2008/01/09/getting-setting-and-removing-tabindex-values-with-javascript/ + // Use proper attribute retrieval(#12072) + var tabindex = jQuery.find.attr( elem, "tabindex" ); + + if ( tabindex ) { + return parseInt( tabindex, 10 ); + } + + if ( + rfocusable.test( elem.nodeName ) || + rclickable.test( elem.nodeName ) && + elem.href + ) { + return 0; + } + + return -1; + } + } + }, + + propFix: { + "for": "htmlFor", + "class": "className" + } +} ); + +// Support: IE <=11 only +// Accessing the selectedIndex property +// forces the browser to respect setting selected +// on the option +// The getter ensures a default option is selected +// when in an optgroup +// eslint rule "no-unused-expressions" is disabled for this code +// since it considers such accessions noop +if ( !support.optSelected ) { + jQuery.propHooks.selected = { + get: function( elem ) { + + /* eslint no-unused-expressions: "off" */ + + var parent = elem.parentNode; + if ( parent && parent.parentNode ) { + parent.parentNode.selectedIndex; + } + return null; + }, + set: function( elem ) { + + /* eslint no-unused-expressions: "off" */ + + var parent = elem.parentNode; + if ( parent ) { + parent.selectedIndex; + + if ( parent.parentNode ) { + parent.parentNode.selectedIndex; + } + } + } + }; +} + +jQuery.each( [ + "tabIndex", + "readOnly", + "maxLength", + "cellSpacing", + "cellPadding", + "rowSpan", + "colSpan", + "useMap", + "frameBorder", + "contentEditable" +], function() { + jQuery.propFix[ this.toLowerCase() ] = this; +} ); + + + + + // Strip and collapse whitespace according to HTML spec + // https://infra.spec.whatwg.org/#strip-and-collapse-ascii-whitespace + function stripAndCollapse( value ) { + var tokens = value.match( rnothtmlwhite ) || []; + return tokens.join( " " ); + } + + +function getClass( elem ) { + return elem.getAttribute && elem.getAttribute( "class" ) || ""; +} + +function classesToArray( value ) { + if ( Array.isArray( value ) ) { + return value; + } + if ( typeof value === "string" ) { + return value.match( rnothtmlwhite ) || []; + } + return []; +} + +jQuery.fn.extend( { + addClass: function( value ) { + var classes, elem, cur, curValue, clazz, j, finalValue, + i = 0; + + if ( isFunction( value ) ) { + return this.each( function( j ) { + jQuery( this ).addClass( value.call( this, j, getClass( this ) ) ); + } ); + } + + classes = classesToArray( value ); + + if ( classes.length ) { + while ( ( elem = this[ i++ ] ) ) { + curValue = getClass( elem ); + cur = elem.nodeType === 1 && ( " " + stripAndCollapse( curValue ) + " " ); + + if ( cur ) { + j = 0; + while ( ( clazz = classes[ j++ ] ) ) { + if ( cur.indexOf( " " + clazz + " " ) < 0 ) { + cur += clazz + " "; + } + } + + // Only assign if different to avoid unneeded rendering. + finalValue = stripAndCollapse( cur ); + if ( curValue !== finalValue ) { + elem.setAttribute( "class", finalValue ); + } + } + } + } + + return this; + }, + + removeClass: function( value ) { + var classes, elem, cur, curValue, clazz, j, finalValue, + i = 0; + + if ( isFunction( value ) ) { + return this.each( function( j ) { + jQuery( this ).removeClass( value.call( this, j, getClass( this ) ) ); + } ); + } + + if ( !arguments.length ) { + return this.attr( "class", "" ); + } + + classes = classesToArray( value ); + + if ( classes.length ) { + while ( ( elem = this[ i++ ] ) ) { + curValue = getClass( elem ); + + // This expression is here for better compressibility (see addClass) + cur = elem.nodeType === 1 && ( " " + stripAndCollapse( curValue ) + " " ); + + if ( cur ) { + j = 0; + while ( ( clazz = classes[ j++ ] ) ) { + + // Remove *all* instances + while ( cur.indexOf( " " + clazz + " " ) > -1 ) { + cur = cur.replace( " " + clazz + " ", " " ); + } + } + + // Only assign if different to avoid unneeded rendering. + finalValue = stripAndCollapse( cur ); + if ( curValue !== finalValue ) { + elem.setAttribute( "class", finalValue ); + } + } + } + } + + return this; + }, + + toggleClass: function( value, stateVal ) { + var type = typeof value, + isValidValue = type === "string" || Array.isArray( value ); + + if ( typeof stateVal === "boolean" && isValidValue ) { + return stateVal ? this.addClass( value ) : this.removeClass( value ); + } + + if ( isFunction( value ) ) { + return this.each( function( i ) { + jQuery( this ).toggleClass( + value.call( this, i, getClass( this ), stateVal ), + stateVal + ); + } ); + } + + return this.each( function() { + var className, i, self, classNames; + + if ( isValidValue ) { + + // Toggle individual class names + i = 0; + self = jQuery( this ); + classNames = classesToArray( value ); + + while ( ( className = classNames[ i++ ] ) ) { + + // Check each className given, space separated list + if ( self.hasClass( className ) ) { + self.removeClass( className ); + } else { + self.addClass( className ); + } + } + + // Toggle whole class name + } else if ( value === undefined || type === "boolean" ) { + className = getClass( this ); + if ( className ) { + + // Store className if set + dataPriv.set( this, "__className__", className ); + } + + // If the element has a class name or if we're passed `false`, + // then remove the whole classname (if there was one, the above saved it). + // Otherwise bring back whatever was previously saved (if anything), + // falling back to the empty string if nothing was stored. + if ( this.setAttribute ) { + this.setAttribute( "class", + className || value === false ? + "" : + dataPriv.get( this, "__className__" ) || "" + ); + } + } + } ); + }, + + hasClass: function( selector ) { + var className, elem, + i = 0; + + className = " " + selector + " "; + while ( ( elem = this[ i++ ] ) ) { + if ( elem.nodeType === 1 && + ( " " + stripAndCollapse( getClass( elem ) ) + " " ).indexOf( className ) > -1 ) { + return true; + } + } + + return false; + } +} ); + + + + +var rreturn = /\r/g; + +jQuery.fn.extend( { + val: function( value ) { + var hooks, ret, valueIsFunction, + elem = this[ 0 ]; + + if ( !arguments.length ) { + if ( elem ) { + hooks = jQuery.valHooks[ elem.type ] || + jQuery.valHooks[ elem.nodeName.toLowerCase() ]; + + if ( hooks && + "get" in hooks && + ( ret = hooks.get( elem, "value" ) ) !== undefined + ) { + return ret; + } + + ret = elem.value; + + // Handle most common string cases + if ( typeof ret === "string" ) { + return ret.replace( rreturn, "" ); + } + + // Handle cases where value is null/undef or number + return ret == null ? "" : ret; + } + + return; + } + + valueIsFunction = isFunction( value ); + + return this.each( function( i ) { + var val; + + if ( this.nodeType !== 1 ) { + return; + } + + if ( valueIsFunction ) { + val = value.call( this, i, jQuery( this ).val() ); + } else { + val = value; + } + + // Treat null/undefined as ""; convert numbers to string + if ( val == null ) { + val = ""; + + } else if ( typeof val === "number" ) { + val += ""; + + } else if ( Array.isArray( val ) ) { + val = jQuery.map( val, function( value ) { + return value == null ? "" : value + ""; + } ); + } + + hooks = jQuery.valHooks[ this.type ] || jQuery.valHooks[ this.nodeName.toLowerCase() ]; + + // If set returns undefined, fall back to normal setting + if ( !hooks || !( "set" in hooks ) || hooks.set( this, val, "value" ) === undefined ) { + this.value = val; + } + } ); + } +} ); + +jQuery.extend( { + valHooks: { + option: { + get: function( elem ) { + + var val = jQuery.find.attr( elem, "value" ); + return val != null ? + val : + + // Support: IE <=10 - 11 only + // option.text throws exceptions (#14686, #14858) + // Strip and collapse whitespace + // https://html.spec.whatwg.org/#strip-and-collapse-whitespace + stripAndCollapse( jQuery.text( elem ) ); + } + }, + select: { + get: function( elem ) { + var value, option, i, + options = elem.options, + index = elem.selectedIndex, + one = elem.type === "select-one", + values = one ? null : [], + max = one ? index + 1 : options.length; + + if ( index < 0 ) { + i = max; + + } else { + i = one ? index : 0; + } + + // Loop through all the selected options + for ( ; i < max; i++ ) { + option = options[ i ]; + + // Support: IE <=9 only + // IE8-9 doesn't update selected after form reset (#2551) + if ( ( option.selected || i === index ) && + + // Don't return options that are disabled or in a disabled optgroup + !option.disabled && + ( !option.parentNode.disabled || + !nodeName( option.parentNode, "optgroup" ) ) ) { + + // Get the specific value for the option + value = jQuery( option ).val(); + + // We don't need an array for one selects + if ( one ) { + return value; + } + + // Multi-Selects return an array + values.push( value ); + } + } + + return values; + }, + + set: function( elem, value ) { + var optionSet, option, + options = elem.options, + values = jQuery.makeArray( value ), + i = options.length; + + while ( i-- ) { + option = options[ i ]; + + /* eslint-disable no-cond-assign */ + + if ( option.selected = + jQuery.inArray( jQuery.valHooks.option.get( option ), values ) > -1 + ) { + optionSet = true; + } + + /* eslint-enable no-cond-assign */ + } + + // Force browsers to behave consistently when non-matching value is set + if ( !optionSet ) { + elem.selectedIndex = -1; + } + return values; + } + } + } +} ); + +// Radios and checkboxes getter/setter +jQuery.each( [ "radio", "checkbox" ], function() { + jQuery.valHooks[ this ] = { + set: function( elem, value ) { + if ( Array.isArray( value ) ) { + return ( elem.checked = jQuery.inArray( jQuery( elem ).val(), value ) > -1 ); + } + } + }; + if ( !support.checkOn ) { + jQuery.valHooks[ this ].get = function( elem ) { + return elem.getAttribute( "value" ) === null ? "on" : elem.value; + }; + } +} ); + + + + +// Return jQuery for attributes-only inclusion + + +support.focusin = "onfocusin" in window; + + +var rfocusMorph = /^(?:focusinfocus|focusoutblur)$/, + stopPropagationCallback = function( e ) { + e.stopPropagation(); + }; + +jQuery.extend( jQuery.event, { + + trigger: function( event, data, elem, onlyHandlers ) { + + var i, cur, tmp, bubbleType, ontype, handle, special, lastElement, + eventPath = [ elem || document ], + type = hasOwn.call( event, "type" ) ? event.type : event, + namespaces = hasOwn.call( event, "namespace" ) ? event.namespace.split( "." ) : []; + + cur = lastElement = tmp = elem = elem || document; + + // Don't do events on text and comment nodes + if ( elem.nodeType === 3 || elem.nodeType === 8 ) { + return; + } + + // focus/blur morphs to focusin/out; ensure we're not firing them right now + if ( rfocusMorph.test( type + jQuery.event.triggered ) ) { + return; + } + + if ( type.indexOf( "." ) > -1 ) { + + // Namespaced trigger; create a regexp to match event type in handle() + namespaces = type.split( "." ); + type = namespaces.shift(); + namespaces.sort(); + } + ontype = type.indexOf( ":" ) < 0 && "on" + type; + + // Caller can pass in a jQuery.Event object, Object, or just an event type string + event = event[ jQuery.expando ] ? + event : + new jQuery.Event( type, typeof event === "object" && event ); + + // Trigger bitmask: & 1 for native handlers; & 2 for jQuery (always true) + event.isTrigger = onlyHandlers ? 2 : 3; + event.namespace = namespaces.join( "." ); + event.rnamespace = event.namespace ? + new RegExp( "(^|\\.)" + namespaces.join( "\\.(?:.*\\.|)" ) + "(\\.|$)" ) : + null; + + // Clean up the event in case it is being reused + event.result = undefined; + if ( !event.target ) { + event.target = elem; + } + + // Clone any incoming data and prepend the event, creating the handler arg list + data = data == null ? + [ event ] : + jQuery.makeArray( data, [ event ] ); + + // Allow special events to draw outside the lines + special = jQuery.event.special[ type ] || {}; + if ( !onlyHandlers && special.trigger && special.trigger.apply( elem, data ) === false ) { + return; + } + + // Determine event propagation path in advance, per W3C events spec (#9951) + // Bubble up to document, then to window; watch for a global ownerDocument var (#9724) + if ( !onlyHandlers && !special.noBubble && !isWindow( elem ) ) { + + bubbleType = special.delegateType || type; + if ( !rfocusMorph.test( bubbleType + type ) ) { + cur = cur.parentNode; + } + for ( ; cur; cur = cur.parentNode ) { + eventPath.push( cur ); + tmp = cur; + } + + // Only add window if we got to document (e.g., not plain obj or detached DOM) + if ( tmp === ( elem.ownerDocument || document ) ) { + eventPath.push( tmp.defaultView || tmp.parentWindow || window ); + } + } + + // Fire handlers on the event path + i = 0; + while ( ( cur = eventPath[ i++ ] ) && !event.isPropagationStopped() ) { + lastElement = cur; + event.type = i > 1 ? + bubbleType : + special.bindType || type; + + // jQuery handler + handle = ( + dataPriv.get( cur, "events" ) || Object.create( null ) + )[ event.type ] && + dataPriv.get( cur, "handle" ); + if ( handle ) { + handle.apply( cur, data ); + } + + // Native handler + handle = ontype && cur[ ontype ]; + if ( handle && handle.apply && acceptData( cur ) ) { + event.result = handle.apply( cur, data ); + if ( event.result === false ) { + event.preventDefault(); + } + } + } + event.type = type; + + // If nobody prevented the default action, do it now + if ( !onlyHandlers && !event.isDefaultPrevented() ) { + + if ( ( !special._default || + special._default.apply( eventPath.pop(), data ) === false ) && + acceptData( elem ) ) { + + // Call a native DOM method on the target with the same name as the event. + // Don't do default actions on window, that's where global variables be (#6170) + if ( ontype && isFunction( elem[ type ] ) && !isWindow( elem ) ) { + + // Don't re-trigger an onFOO event when we call its FOO() method + tmp = elem[ ontype ]; + + if ( tmp ) { + elem[ ontype ] = null; + } + + // Prevent re-triggering of the same event, since we already bubbled it above + jQuery.event.triggered = type; + + if ( event.isPropagationStopped() ) { + lastElement.addEventListener( type, stopPropagationCallback ); + } + + elem[ type ](); + + if ( event.isPropagationStopped() ) { + lastElement.removeEventListener( type, stopPropagationCallback ); + } + + jQuery.event.triggered = undefined; + + if ( tmp ) { + elem[ ontype ] = tmp; + } + } + } + } + + return event.result; + }, + + // Piggyback on a donor event to simulate a different one + // Used only for `focus(in | out)` events + simulate: function( type, elem, event ) { + var e = jQuery.extend( + new jQuery.Event(), + event, + { + type: type, + isSimulated: true + } + ); + + jQuery.event.trigger( e, null, elem ); + } + +} ); + +jQuery.fn.extend( { + + trigger: function( type, data ) { + return this.each( function() { + jQuery.event.trigger( type, data, this ); + } ); + }, + triggerHandler: function( type, data ) { + var elem = this[ 0 ]; + if ( elem ) { + return jQuery.event.trigger( type, data, elem, true ); + } + } +} ); + + +// Support: Firefox <=44 +// Firefox doesn't have focus(in | out) events +// Related ticket - https://bugzilla.mozilla.org/show_bug.cgi?id=687787 +// +// Support: Chrome <=48 - 49, Safari <=9.0 - 9.1 +// focus(in | out) events fire after focus & blur events, +// which is spec violation - http://www.w3.org/TR/DOM-Level-3-Events/#events-focusevent-event-order +// Related ticket - https://bugs.chromium.org/p/chromium/issues/detail?id=449857 +if ( !support.focusin ) { + jQuery.each( { focus: "focusin", blur: "focusout" }, function( orig, fix ) { + + // Attach a single capturing handler on the document while someone wants focusin/focusout + var handler = function( event ) { + jQuery.event.simulate( fix, event.target, jQuery.event.fix( event ) ); + }; + + jQuery.event.special[ fix ] = { + setup: function() { + + // Handle: regular nodes (via `this.ownerDocument`), window + // (via `this.document`) & document (via `this`). + var doc = this.ownerDocument || this.document || this, + attaches = dataPriv.access( doc, fix ); + + if ( !attaches ) { + doc.addEventListener( orig, handler, true ); + } + dataPriv.access( doc, fix, ( attaches || 0 ) + 1 ); + }, + teardown: function() { + var doc = this.ownerDocument || this.document || this, + attaches = dataPriv.access( doc, fix ) - 1; + + if ( !attaches ) { + doc.removeEventListener( orig, handler, true ); + dataPriv.remove( doc, fix ); + + } else { + dataPriv.access( doc, fix, attaches ); + } + } + }; + } ); +} +var location = window.location; + +var nonce = { guid: Date.now() }; + +var rquery = ( /\?/ ); + + + +// Cross-browser xml parsing +jQuery.parseXML = function( data ) { + var xml; + if ( !data || typeof data !== "string" ) { + return null; + } + + // Support: IE 9 - 11 only + // IE throws on parseFromString with invalid input. + try { + xml = ( new window.DOMParser() ).parseFromString( data, "text/xml" ); + } catch ( e ) { + xml = undefined; + } + + if ( !xml || xml.getElementsByTagName( "parsererror" ).length ) { + jQuery.error( "Invalid XML: " + data ); + } + return xml; +}; + + +var + rbracket = /\[\]$/, + rCRLF = /\r?\n/g, + rsubmitterTypes = /^(?:submit|button|image|reset|file)$/i, + rsubmittable = /^(?:input|select|textarea|keygen)/i; + +function buildParams( prefix, obj, traditional, add ) { + var name; + + if ( Array.isArray( obj ) ) { + + // Serialize array item. + jQuery.each( obj, function( i, v ) { + if ( traditional || rbracket.test( prefix ) ) { + + // Treat each array item as a scalar. + add( prefix, v ); + + } else { + + // Item is non-scalar (array or object), encode its numeric index. + buildParams( + prefix + "[" + ( typeof v === "object" && v != null ? i : "" ) + "]", + v, + traditional, + add + ); + } + } ); + + } else if ( !traditional && toType( obj ) === "object" ) { + + // Serialize object item. + for ( name in obj ) { + buildParams( prefix + "[" + name + "]", obj[ name ], traditional, add ); + } + + } else { + + // Serialize scalar item. + add( prefix, obj ); + } +} + +// Serialize an array of form elements or a set of +// key/values into a query string +jQuery.param = function( a, traditional ) { + var prefix, + s = [], + add = function( key, valueOrFunction ) { + + // If value is a function, invoke it and use its return value + var value = isFunction( valueOrFunction ) ? + valueOrFunction() : + valueOrFunction; + + s[ s.length ] = encodeURIComponent( key ) + "=" + + encodeURIComponent( value == null ? "" : value ); + }; + + if ( a == null ) { + return ""; + } + + // If an array was passed in, assume that it is an array of form elements. + if ( Array.isArray( a ) || ( a.jquery && !jQuery.isPlainObject( a ) ) ) { + + // Serialize the form elements + jQuery.each( a, function() { + add( this.name, this.value ); + } ); + + } else { + + // If traditional, encode the "old" way (the way 1.3.2 or older + // did it), otherwise encode params recursively. + for ( prefix in a ) { + buildParams( prefix, a[ prefix ], traditional, add ); + } + } + + // Return the resulting serialization + return s.join( "&" ); +}; + +jQuery.fn.extend( { + serialize: function() { + return jQuery.param( this.serializeArray() ); + }, + serializeArray: function() { + return this.map( function() { + + // Can add propHook for "elements" to filter or add form elements + var elements = jQuery.prop( this, "elements" ); + return elements ? jQuery.makeArray( elements ) : this; + } ) + .filter( function() { + var type = this.type; + + // Use .is( ":disabled" ) so that fieldset[disabled] works + return this.name && !jQuery( this ).is( ":disabled" ) && + rsubmittable.test( this.nodeName ) && !rsubmitterTypes.test( type ) && + ( this.checked || !rcheckableType.test( type ) ); + } ) + .map( function( _i, elem ) { + var val = jQuery( this ).val(); + + if ( val == null ) { + return null; + } + + if ( Array.isArray( val ) ) { + return jQuery.map( val, function( val ) { + return { name: elem.name, value: val.replace( rCRLF, "\r\n" ) }; + } ); + } + + return { name: elem.name, value: val.replace( rCRLF, "\r\n" ) }; + } ).get(); + } +} ); + + +var + r20 = /%20/g, + rhash = /#.*$/, + rantiCache = /([?&])_=[^&]*/, + rheaders = /^(.*?):[ \t]*([^\r\n]*)$/mg, + + // #7653, #8125, #8152: local protocol detection + rlocalProtocol = /^(?:about|app|app-storage|.+-extension|file|res|widget):$/, + rnoContent = /^(?:GET|HEAD)$/, + rprotocol = /^\/\//, + + /* Prefilters + * 1) They are useful to introduce custom dataTypes (see ajax/jsonp.js for an example) + * 2) These are called: + * - BEFORE asking for a transport + * - AFTER param serialization (s.data is a string if s.processData is true) + * 3) key is the dataType + * 4) the catchall symbol "*" can be used + * 5) execution will start with transport dataType and THEN continue down to "*" if needed + */ + prefilters = {}, + + /* Transports bindings + * 1) key is the dataType + * 2) the catchall symbol "*" can be used + * 3) selection will start with transport dataType and THEN go to "*" if needed + */ + transports = {}, + + // Avoid comment-prolog char sequence (#10098); must appease lint and evade compression + allTypes = "*/".concat( "*" ), + + // Anchor tag for parsing the document origin + originAnchor = document.createElement( "a" ); + originAnchor.href = location.href; + +// Base "constructor" for jQuery.ajaxPrefilter and jQuery.ajaxTransport +function addToPrefiltersOrTransports( structure ) { + + // dataTypeExpression is optional and defaults to "*" + return function( dataTypeExpression, func ) { + + if ( typeof dataTypeExpression !== "string" ) { + func = dataTypeExpression; + dataTypeExpression = "*"; + } + + var dataType, + i = 0, + dataTypes = dataTypeExpression.toLowerCase().match( rnothtmlwhite ) || []; + + if ( isFunction( func ) ) { + + // For each dataType in the dataTypeExpression + while ( ( dataType = dataTypes[ i++ ] ) ) { + + // Prepend if requested + if ( dataType[ 0 ] === "+" ) { + dataType = dataType.slice( 1 ) || "*"; + ( structure[ dataType ] = structure[ dataType ] || [] ).unshift( func ); + + // Otherwise append + } else { + ( structure[ dataType ] = structure[ dataType ] || [] ).push( func ); + } + } + } + }; +} + +// Base inspection function for prefilters and transports +function inspectPrefiltersOrTransports( structure, options, originalOptions, jqXHR ) { + + var inspected = {}, + seekingTransport = ( structure === transports ); + + function inspect( dataType ) { + var selected; + inspected[ dataType ] = true; + jQuery.each( structure[ dataType ] || [], function( _, prefilterOrFactory ) { + var dataTypeOrTransport = prefilterOrFactory( options, originalOptions, jqXHR ); + if ( typeof dataTypeOrTransport === "string" && + !seekingTransport && !inspected[ dataTypeOrTransport ] ) { + + options.dataTypes.unshift( dataTypeOrTransport ); + inspect( dataTypeOrTransport ); + return false; + } else if ( seekingTransport ) { + return !( selected = dataTypeOrTransport ); + } + } ); + return selected; + } + + return inspect( options.dataTypes[ 0 ] ) || !inspected[ "*" ] && inspect( "*" ); +} + +// A special extend for ajax options +// that takes "flat" options (not to be deep extended) +// Fixes #9887 +function ajaxExtend( target, src ) { + var key, deep, + flatOptions = jQuery.ajaxSettings.flatOptions || {}; + + for ( key in src ) { + if ( src[ key ] !== undefined ) { + ( flatOptions[ key ] ? target : ( deep || ( deep = {} ) ) )[ key ] = src[ key ]; + } + } + if ( deep ) { + jQuery.extend( true, target, deep ); + } + + return target; +} + +/* Handles responses to an ajax request: + * - finds the right dataType (mediates between content-type and expected dataType) + * - returns the corresponding response + */ +function ajaxHandleResponses( s, jqXHR, responses ) { + + var ct, type, finalDataType, firstDataType, + contents = s.contents, + dataTypes = s.dataTypes; + + // Remove auto dataType and get content-type in the process + while ( dataTypes[ 0 ] === "*" ) { + dataTypes.shift(); + if ( ct === undefined ) { + ct = s.mimeType || jqXHR.getResponseHeader( "Content-Type" ); + } + } + + // Check if we're dealing with a known content-type + if ( ct ) { + for ( type in contents ) { + if ( contents[ type ] && contents[ type ].test( ct ) ) { + dataTypes.unshift( type ); + break; + } + } + } + + // Check to see if we have a response for the expected dataType + if ( dataTypes[ 0 ] in responses ) { + finalDataType = dataTypes[ 0 ]; + } else { + + // Try convertible dataTypes + for ( type in responses ) { + if ( !dataTypes[ 0 ] || s.converters[ type + " " + dataTypes[ 0 ] ] ) { + finalDataType = type; + break; + } + if ( !firstDataType ) { + firstDataType = type; + } + } + + // Or just use first one + finalDataType = finalDataType || firstDataType; + } + + // If we found a dataType + // We add the dataType to the list if needed + // and return the corresponding response + if ( finalDataType ) { + if ( finalDataType !== dataTypes[ 0 ] ) { + dataTypes.unshift( finalDataType ); + } + return responses[ finalDataType ]; + } +} + +/* Chain conversions given the request and the original response + * Also sets the responseXXX fields on the jqXHR instance + */ +function ajaxConvert( s, response, jqXHR, isSuccess ) { + var conv2, current, conv, tmp, prev, + converters = {}, + + // Work with a copy of dataTypes in case we need to modify it for conversion + dataTypes = s.dataTypes.slice(); + + // Create converters map with lowercased keys + if ( dataTypes[ 1 ] ) { + for ( conv in s.converters ) { + converters[ conv.toLowerCase() ] = s.converters[ conv ]; + } + } + + current = dataTypes.shift(); + + // Convert to each sequential dataType + while ( current ) { + + if ( s.responseFields[ current ] ) { + jqXHR[ s.responseFields[ current ] ] = response; + } + + // Apply the dataFilter if provided + if ( !prev && isSuccess && s.dataFilter ) { + response = s.dataFilter( response, s.dataType ); + } + + prev = current; + current = dataTypes.shift(); + + if ( current ) { + + // There's only work to do if current dataType is non-auto + if ( current === "*" ) { + + current = prev; + + // Convert response if prev dataType is non-auto and differs from current + } else if ( prev !== "*" && prev !== current ) { + + // Seek a direct converter + conv = converters[ prev + " " + current ] || converters[ "* " + current ]; + + // If none found, seek a pair + if ( !conv ) { + for ( conv2 in converters ) { + + // If conv2 outputs current + tmp = conv2.split( " " ); + if ( tmp[ 1 ] === current ) { + + // If prev can be converted to accepted input + conv = converters[ prev + " " + tmp[ 0 ] ] || + converters[ "* " + tmp[ 0 ] ]; + if ( conv ) { + + // Condense equivalence converters + if ( conv === true ) { + conv = converters[ conv2 ]; + + // Otherwise, insert the intermediate dataType + } else if ( converters[ conv2 ] !== true ) { + current = tmp[ 0 ]; + dataTypes.unshift( tmp[ 1 ] ); + } + break; + } + } + } + } + + // Apply converter (if not an equivalence) + if ( conv !== true ) { + + // Unless errors are allowed to bubble, catch and return them + if ( conv && s.throws ) { + response = conv( response ); + } else { + try { + response = conv( response ); + } catch ( e ) { + return { + state: "parsererror", + error: conv ? e : "No conversion from " + prev + " to " + current + }; + } + } + } + } + } + } + + return { state: "success", data: response }; +} + +jQuery.extend( { + + // Counter for holding the number of active queries + active: 0, + + // Last-Modified header cache for next request + lastModified: {}, + etag: {}, + + ajaxSettings: { + url: location.href, + type: "GET", + isLocal: rlocalProtocol.test( location.protocol ), + global: true, + processData: true, + async: true, + contentType: "application/x-www-form-urlencoded; charset=UTF-8", + + /* + timeout: 0, + data: null, + dataType: null, + username: null, + password: null, + cache: null, + throws: false, + traditional: false, + headers: {}, + */ + + accepts: { + "*": allTypes, + text: "text/plain", + html: "text/html", + xml: "application/xml, text/xml", + json: "application/json, text/javascript" + }, + + contents: { + xml: /\bxml\b/, + html: /\bhtml/, + json: /\bjson\b/ + }, + + responseFields: { + xml: "responseXML", + text: "responseText", + json: "responseJSON" + }, + + // Data converters + // Keys separate source (or catchall "*") and destination types with a single space + converters: { + + // Convert anything to text + "* text": String, + + // Text to html (true = no transformation) + "text html": true, + + // Evaluate text as a json expression + "text json": JSON.parse, + + // Parse text as xml + "text xml": jQuery.parseXML + }, + + // For options that shouldn't be deep extended: + // you can add your own custom options here if + // and when you create one that shouldn't be + // deep extended (see ajaxExtend) + flatOptions: { + url: true, + context: true + } + }, + + // Creates a full fledged settings object into target + // with both ajaxSettings and settings fields. + // If target is omitted, writes into ajaxSettings. + ajaxSetup: function( target, settings ) { + return settings ? + + // Building a settings object + ajaxExtend( ajaxExtend( target, jQuery.ajaxSettings ), settings ) : + + // Extending ajaxSettings + ajaxExtend( jQuery.ajaxSettings, target ); + }, + + ajaxPrefilter: addToPrefiltersOrTransports( prefilters ), + ajaxTransport: addToPrefiltersOrTransports( transports ), + + // Main method + ajax: function( url, options ) { + + // If url is an object, simulate pre-1.5 signature + if ( typeof url === "object" ) { + options = url; + url = undefined; + } + + // Force options to be an object + options = options || {}; + + var transport, + + // URL without anti-cache param + cacheURL, + + // Response headers + responseHeadersString, + responseHeaders, + + // timeout handle + timeoutTimer, + + // Url cleanup var + urlAnchor, + + // Request state (becomes false upon send and true upon completion) + completed, + + // To know if global events are to be dispatched + fireGlobals, + + // Loop variable + i, + + // uncached part of the url + uncached, + + // Create the final options object + s = jQuery.ajaxSetup( {}, options ), + + // Callbacks context + callbackContext = s.context || s, + + // Context for global events is callbackContext if it is a DOM node or jQuery collection + globalEventContext = s.context && + ( callbackContext.nodeType || callbackContext.jquery ) ? + jQuery( callbackContext ) : + jQuery.event, + + // Deferreds + deferred = jQuery.Deferred(), + completeDeferred = jQuery.Callbacks( "once memory" ), + + // Status-dependent callbacks + statusCode = s.statusCode || {}, + + // Headers (they are sent all at once) + requestHeaders = {}, + requestHeadersNames = {}, + + // Default abort message + strAbort = "canceled", + + // Fake xhr + jqXHR = { + readyState: 0, + + // Builds headers hashtable if needed + getResponseHeader: function( key ) { + var match; + if ( completed ) { + if ( !responseHeaders ) { + responseHeaders = {}; + while ( ( match = rheaders.exec( responseHeadersString ) ) ) { + responseHeaders[ match[ 1 ].toLowerCase() + " " ] = + ( responseHeaders[ match[ 1 ].toLowerCase() + " " ] || [] ) + .concat( match[ 2 ] ); + } + } + match = responseHeaders[ key.toLowerCase() + " " ]; + } + return match == null ? null : match.join( ", " ); + }, + + // Raw string + getAllResponseHeaders: function() { + return completed ? responseHeadersString : null; + }, + + // Caches the header + setRequestHeader: function( name, value ) { + if ( completed == null ) { + name = requestHeadersNames[ name.toLowerCase() ] = + requestHeadersNames[ name.toLowerCase() ] || name; + requestHeaders[ name ] = value; + } + return this; + }, + + // Overrides response content-type header + overrideMimeType: function( type ) { + if ( completed == null ) { + s.mimeType = type; + } + return this; + }, + + // Status-dependent callbacks + statusCode: function( map ) { + var code; + if ( map ) { + if ( completed ) { + + // Execute the appropriate callbacks + jqXHR.always( map[ jqXHR.status ] ); + } else { + + // Lazy-add the new callbacks in a way that preserves old ones + for ( code in map ) { + statusCode[ code ] = [ statusCode[ code ], map[ code ] ]; + } + } + } + return this; + }, + + // Cancel the request + abort: function( statusText ) { + var finalText = statusText || strAbort; + if ( transport ) { + transport.abort( finalText ); + } + done( 0, finalText ); + return this; + } + }; + + // Attach deferreds + deferred.promise( jqXHR ); + + // Add protocol if not provided (prefilters might expect it) + // Handle falsy url in the settings object (#10093: consistency with old signature) + // We also use the url parameter if available + s.url = ( ( url || s.url || location.href ) + "" ) + .replace( rprotocol, location.protocol + "//" ); + + // Alias method option to type as per ticket #12004 + s.type = options.method || options.type || s.method || s.type; + + // Extract dataTypes list + s.dataTypes = ( s.dataType || "*" ).toLowerCase().match( rnothtmlwhite ) || [ "" ]; + + // A cross-domain request is in order when the origin doesn't match the current origin. + if ( s.crossDomain == null ) { + urlAnchor = document.createElement( "a" ); + + // Support: IE <=8 - 11, Edge 12 - 15 + // IE throws exception on accessing the href property if url is malformed, + // e.g. http://example.com:80x/ + try { + urlAnchor.href = s.url; + + // Support: IE <=8 - 11 only + // Anchor's host property isn't correctly set when s.url is relative + urlAnchor.href = urlAnchor.href; + s.crossDomain = originAnchor.protocol + "//" + originAnchor.host !== + urlAnchor.protocol + "//" + urlAnchor.host; + } catch ( e ) { + + // If there is an error parsing the URL, assume it is crossDomain, + // it can be rejected by the transport if it is invalid + s.crossDomain = true; + } + } + + // Convert data if not already a string + if ( s.data && s.processData && typeof s.data !== "string" ) { + s.data = jQuery.param( s.data, s.traditional ); + } + + // Apply prefilters + inspectPrefiltersOrTransports( prefilters, s, options, jqXHR ); + + // If request was aborted inside a prefilter, stop there + if ( completed ) { + return jqXHR; + } + + // We can fire global events as of now if asked to + // Don't fire events if jQuery.event is undefined in an AMD-usage scenario (#15118) + fireGlobals = jQuery.event && s.global; + + // Watch for a new set of requests + if ( fireGlobals && jQuery.active++ === 0 ) { + jQuery.event.trigger( "ajaxStart" ); + } + + // Uppercase the type + s.type = s.type.toUpperCase(); + + // Determine if request has content + s.hasContent = !rnoContent.test( s.type ); + + // Save the URL in case we're toying with the If-Modified-Since + // and/or If-None-Match header later on + // Remove hash to simplify url manipulation + cacheURL = s.url.replace( rhash, "" ); + + // More options handling for requests with no content + if ( !s.hasContent ) { + + // Remember the hash so we can put it back + uncached = s.url.slice( cacheURL.length ); + + // If data is available and should be processed, append data to url + if ( s.data && ( s.processData || typeof s.data === "string" ) ) { + cacheURL += ( rquery.test( cacheURL ) ? "&" : "?" ) + s.data; + + // #9682: remove data so that it's not used in an eventual retry + delete s.data; + } + + // Add or update anti-cache param if needed + if ( s.cache === false ) { + cacheURL = cacheURL.replace( rantiCache, "$1" ); + uncached = ( rquery.test( cacheURL ) ? "&" : "?" ) + "_=" + ( nonce.guid++ ) + + uncached; + } + + // Put hash and anti-cache on the URL that will be requested (gh-1732) + s.url = cacheURL + uncached; + + // Change '%20' to '+' if this is encoded form body content (gh-2658) + } else if ( s.data && s.processData && + ( s.contentType || "" ).indexOf( "application/x-www-form-urlencoded" ) === 0 ) { + s.data = s.data.replace( r20, "+" ); + } + + // Set the If-Modified-Since and/or If-None-Match header, if in ifModified mode. + if ( s.ifModified ) { + if ( jQuery.lastModified[ cacheURL ] ) { + jqXHR.setRequestHeader( "If-Modified-Since", jQuery.lastModified[ cacheURL ] ); + } + if ( jQuery.etag[ cacheURL ] ) { + jqXHR.setRequestHeader( "If-None-Match", jQuery.etag[ cacheURL ] ); + } + } + + // Set the correct header, if data is being sent + if ( s.data && s.hasContent && s.contentType !== false || options.contentType ) { + jqXHR.setRequestHeader( "Content-Type", s.contentType ); + } + + // Set the Accepts header for the server, depending on the dataType + jqXHR.setRequestHeader( + "Accept", + s.dataTypes[ 0 ] && s.accepts[ s.dataTypes[ 0 ] ] ? + s.accepts[ s.dataTypes[ 0 ] ] + + ( s.dataTypes[ 0 ] !== "*" ? ", " + allTypes + "; q=0.01" : "" ) : + s.accepts[ "*" ] + ); + + // Check for headers option + for ( i in s.headers ) { + jqXHR.setRequestHeader( i, s.headers[ i ] ); + } + + // Allow custom headers/mimetypes and early abort + if ( s.beforeSend && + ( s.beforeSend.call( callbackContext, jqXHR, s ) === false || completed ) ) { + + // Abort if not done already and return + return jqXHR.abort(); + } + + // Aborting is no longer a cancellation + strAbort = "abort"; + + // Install callbacks on deferreds + completeDeferred.add( s.complete ); + jqXHR.done( s.success ); + jqXHR.fail( s.error ); + + // Get transport + transport = inspectPrefiltersOrTransports( transports, s, options, jqXHR ); + + // If no transport, we auto-abort + if ( !transport ) { + done( -1, "No Transport" ); + } else { + jqXHR.readyState = 1; + + // Send global event + if ( fireGlobals ) { + globalEventContext.trigger( "ajaxSend", [ jqXHR, s ] ); + } + + // If request was aborted inside ajaxSend, stop there + if ( completed ) { + return jqXHR; + } + + // Timeout + if ( s.async && s.timeout > 0 ) { + timeoutTimer = window.setTimeout( function() { + jqXHR.abort( "timeout" ); + }, s.timeout ); + } + + try { + completed = false; + transport.send( requestHeaders, done ); + } catch ( e ) { + + // Rethrow post-completion exceptions + if ( completed ) { + throw e; + } + + // Propagate others as results + done( -1, e ); + } + } + + // Callback for when everything is done + function done( status, nativeStatusText, responses, headers ) { + var isSuccess, success, error, response, modified, + statusText = nativeStatusText; + + // Ignore repeat invocations + if ( completed ) { + return; + } + + completed = true; + + // Clear timeout if it exists + if ( timeoutTimer ) { + window.clearTimeout( timeoutTimer ); + } + + // Dereference transport for early garbage collection + // (no matter how long the jqXHR object will be used) + transport = undefined; + + // Cache response headers + responseHeadersString = headers || ""; + + // Set readyState + jqXHR.readyState = status > 0 ? 4 : 0; + + // Determine if successful + isSuccess = status >= 200 && status < 300 || status === 304; + + // Get response data + if ( responses ) { + response = ajaxHandleResponses( s, jqXHR, responses ); + } + + // Use a noop converter for missing script + if ( !isSuccess && jQuery.inArray( "script", s.dataTypes ) > -1 ) { + s.converters[ "text script" ] = function() {}; + } + + // Convert no matter what (that way responseXXX fields are always set) + response = ajaxConvert( s, response, jqXHR, isSuccess ); + + // If successful, handle type chaining + if ( isSuccess ) { + + // Set the If-Modified-Since and/or If-None-Match header, if in ifModified mode. + if ( s.ifModified ) { + modified = jqXHR.getResponseHeader( "Last-Modified" ); + if ( modified ) { + jQuery.lastModified[ cacheURL ] = modified; + } + modified = jqXHR.getResponseHeader( "etag" ); + if ( modified ) { + jQuery.etag[ cacheURL ] = modified; + } + } + + // if no content + if ( status === 204 || s.type === "HEAD" ) { + statusText = "nocontent"; + + // if not modified + } else if ( status === 304 ) { + statusText = "notmodified"; + + // If we have data, let's convert it + } else { + statusText = response.state; + success = response.data; + error = response.error; + isSuccess = !error; + } + } else { + + // Extract error from statusText and normalize for non-aborts + error = statusText; + if ( status || !statusText ) { + statusText = "error"; + if ( status < 0 ) { + status = 0; + } + } + } + + // Set data for the fake xhr object + jqXHR.status = status; + jqXHR.statusText = ( nativeStatusText || statusText ) + ""; + + // Success/Error + if ( isSuccess ) { + deferred.resolveWith( callbackContext, [ success, statusText, jqXHR ] ); + } else { + deferred.rejectWith( callbackContext, [ jqXHR, statusText, error ] ); + } + + // Status-dependent callbacks + jqXHR.statusCode( statusCode ); + statusCode = undefined; + + if ( fireGlobals ) { + globalEventContext.trigger( isSuccess ? "ajaxSuccess" : "ajaxError", + [ jqXHR, s, isSuccess ? success : error ] ); + } + + // Complete + completeDeferred.fireWith( callbackContext, [ jqXHR, statusText ] ); + + if ( fireGlobals ) { + globalEventContext.trigger( "ajaxComplete", [ jqXHR, s ] ); + + // Handle the global AJAX counter + if ( !( --jQuery.active ) ) { + jQuery.event.trigger( "ajaxStop" ); + } + } + } + + return jqXHR; + }, + + getJSON: function( url, data, callback ) { + return jQuery.get( url, data, callback, "json" ); + }, + + getScript: function( url, callback ) { + return jQuery.get( url, undefined, callback, "script" ); + } +} ); + +jQuery.each( [ "get", "post" ], function( _i, method ) { + jQuery[ method ] = function( url, data, callback, type ) { + + // Shift arguments if data argument was omitted + if ( isFunction( data ) ) { + type = type || callback; + callback = data; + data = undefined; + } + + // The url can be an options object (which then must have .url) + return jQuery.ajax( jQuery.extend( { + url: url, + type: method, + dataType: type, + data: data, + success: callback + }, jQuery.isPlainObject( url ) && url ) ); + }; +} ); + +jQuery.ajaxPrefilter( function( s ) { + var i; + for ( i in s.headers ) { + if ( i.toLowerCase() === "content-type" ) { + s.contentType = s.headers[ i ] || ""; + } + } +} ); + + +jQuery._evalUrl = function( url, options, doc ) { + return jQuery.ajax( { + url: url, + + // Make this explicit, since user can override this through ajaxSetup (#11264) + type: "GET", + dataType: "script", + cache: true, + async: false, + global: false, + + // Only evaluate the response if it is successful (gh-4126) + // dataFilter is not invoked for failure responses, so using it instead + // of the default converter is kludgy but it works. + converters: { + "text script": function() {} + }, + dataFilter: function( response ) { + jQuery.globalEval( response, options, doc ); + } + } ); +}; + + +jQuery.fn.extend( { + wrapAll: function( html ) { + var wrap; + + if ( this[ 0 ] ) { + if ( isFunction( html ) ) { + html = html.call( this[ 0 ] ); + } + + // The elements to wrap the target around + wrap = jQuery( html, this[ 0 ].ownerDocument ).eq( 0 ).clone( true ); + + if ( this[ 0 ].parentNode ) { + wrap.insertBefore( this[ 0 ] ); + } + + wrap.map( function() { + var elem = this; + + while ( elem.firstElementChild ) { + elem = elem.firstElementChild; + } + + return elem; + } ).append( this ); + } + + return this; + }, + + wrapInner: function( html ) { + if ( isFunction( html ) ) { + return this.each( function( i ) { + jQuery( this ).wrapInner( html.call( this, i ) ); + } ); + } + + return this.each( function() { + var self = jQuery( this ), + contents = self.contents(); + + if ( contents.length ) { + contents.wrapAll( html ); + + } else { + self.append( html ); + } + } ); + }, + + wrap: function( html ) { + var htmlIsFunction = isFunction( html ); + + return this.each( function( i ) { + jQuery( this ).wrapAll( htmlIsFunction ? html.call( this, i ) : html ); + } ); + }, + + unwrap: function( selector ) { + this.parent( selector ).not( "body" ).each( function() { + jQuery( this ).replaceWith( this.childNodes ); + } ); + return this; + } +} ); + + +jQuery.expr.pseudos.hidden = function( elem ) { + return !jQuery.expr.pseudos.visible( elem ); +}; +jQuery.expr.pseudos.visible = function( elem ) { + return !!( elem.offsetWidth || elem.offsetHeight || elem.getClientRects().length ); +}; + + + + +jQuery.ajaxSettings.xhr = function() { + try { + return new window.XMLHttpRequest(); + } catch ( e ) {} +}; + +var xhrSuccessStatus = { + + // File protocol always yields status code 0, assume 200 + 0: 200, + + // Support: IE <=9 only + // #1450: sometimes IE returns 1223 when it should be 204 + 1223: 204 + }, + xhrSupported = jQuery.ajaxSettings.xhr(); + +support.cors = !!xhrSupported && ( "withCredentials" in xhrSupported ); +support.ajax = xhrSupported = !!xhrSupported; + +jQuery.ajaxTransport( function( options ) { + var callback, errorCallback; + + // Cross domain only allowed if supported through XMLHttpRequest + if ( support.cors || xhrSupported && !options.crossDomain ) { + return { + send: function( headers, complete ) { + var i, + xhr = options.xhr(); + + xhr.open( + options.type, + options.url, + options.async, + options.username, + options.password + ); + + // Apply custom fields if provided + if ( options.xhrFields ) { + for ( i in options.xhrFields ) { + xhr[ i ] = options.xhrFields[ i ]; + } + } + + // Override mime type if needed + if ( options.mimeType && xhr.overrideMimeType ) { + xhr.overrideMimeType( options.mimeType ); + } + + // X-Requested-With header + // For cross-domain requests, seeing as conditions for a preflight are + // akin to a jigsaw puzzle, we simply never set it to be sure. + // (it can always be set on a per-request basis or even using ajaxSetup) + // For same-domain requests, won't change header if already provided. + if ( !options.crossDomain && !headers[ "X-Requested-With" ] ) { + headers[ "X-Requested-With" ] = "XMLHttpRequest"; + } + + // Set headers + for ( i in headers ) { + xhr.setRequestHeader( i, headers[ i ] ); + } + + // Callback + callback = function( type ) { + return function() { + if ( callback ) { + callback = errorCallback = xhr.onload = + xhr.onerror = xhr.onabort = xhr.ontimeout = + xhr.onreadystatechange = null; + + if ( type === "abort" ) { + xhr.abort(); + } else if ( type === "error" ) { + + // Support: IE <=9 only + // On a manual native abort, IE9 throws + // errors on any property access that is not readyState + if ( typeof xhr.status !== "number" ) { + complete( 0, "error" ); + } else { + complete( + + // File: protocol always yields status 0; see #8605, #14207 + xhr.status, + xhr.statusText + ); + } + } else { + complete( + xhrSuccessStatus[ xhr.status ] || xhr.status, + xhr.statusText, + + // Support: IE <=9 only + // IE9 has no XHR2 but throws on binary (trac-11426) + // For XHR2 non-text, let the caller handle it (gh-2498) + ( xhr.responseType || "text" ) !== "text" || + typeof xhr.responseText !== "string" ? + { binary: xhr.response } : + { text: xhr.responseText }, + xhr.getAllResponseHeaders() + ); + } + } + }; + }; + + // Listen to events + xhr.onload = callback(); + errorCallback = xhr.onerror = xhr.ontimeout = callback( "error" ); + + // Support: IE 9 only + // Use onreadystatechange to replace onabort + // to handle uncaught aborts + if ( xhr.onabort !== undefined ) { + xhr.onabort = errorCallback; + } else { + xhr.onreadystatechange = function() { + + // Check readyState before timeout as it changes + if ( xhr.readyState === 4 ) { + + // Allow onerror to be called first, + // but that will not handle a native abort + // Also, save errorCallback to a variable + // as xhr.onerror cannot be accessed + window.setTimeout( function() { + if ( callback ) { + errorCallback(); + } + } ); + } + }; + } + + // Create the abort callback + callback = callback( "abort" ); + + try { + + // Do send the request (this may raise an exception) + xhr.send( options.hasContent && options.data || null ); + } catch ( e ) { + + // #14683: Only rethrow if this hasn't been notified as an error yet + if ( callback ) { + throw e; + } + } + }, + + abort: function() { + if ( callback ) { + callback(); + } + } + }; + } +} ); + + + + +// Prevent auto-execution of scripts when no explicit dataType was provided (See gh-2432) +jQuery.ajaxPrefilter( function( s ) { + if ( s.crossDomain ) { + s.contents.script = false; + } +} ); + +// Install script dataType +jQuery.ajaxSetup( { + accepts: { + script: "text/javascript, application/javascript, " + + "application/ecmascript, application/x-ecmascript" + }, + contents: { + script: /\b(?:java|ecma)script\b/ + }, + converters: { + "text script": function( text ) { + jQuery.globalEval( text ); + return text; + } + } +} ); + +// Handle cache's special case and crossDomain +jQuery.ajaxPrefilter( "script", function( s ) { + if ( s.cache === undefined ) { + s.cache = false; + } + if ( s.crossDomain ) { + s.type = "GET"; + } +} ); + +// Bind script tag hack transport +jQuery.ajaxTransport( "script", function( s ) { + + // This transport only deals with cross domain or forced-by-attrs requests + if ( s.crossDomain || s.scriptAttrs ) { + var script, callback; + return { + send: function( _, complete ) { + script = jQuery( " +{% endmacro %} \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/chapter1.html b/doc/LectureNotes/_build/html/chapter1.html new file mode 100644 index 000000000..748370b8c --- /dev/null +++ b/doc/LectureNotes/_build/html/chapter1.html @@ -0,0 +1,2890 @@ + + + + + + + + 1. Linear Regression, basic Elements — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ + +
+
+ +
+ +
+

1. Linear Regression, basic Elements

+

Video of Lecture

+
+

1.1. Introduction

+

Our emphasis throughout this series of lectures
+is on understanding the mathematical aspects of +different algorithms used in the fields of data analysis and machine learning.

+

However, where possible we will emphasize the +importance of using available software. We start thus with a hands-on +and top-down approach to machine learning. The aim is thus to start with +relevant data or data we have produced +and use these to introduce statistical data analysis +concepts and machine learning algorithms before we delve into the +algorithms themselves. The examples we will use in the beginning, start with simple +polynomials with random noise added. We will use the Python +software package Scikit-Learn and +introduce various machine learning algorithms to make fits of +the data and predictions. We move thereafter to more interesting +cases such as data from say experiments (below we will look at experimental nuclear binding energies as an example). +These are examples where we can easily set up the data and +then use machine learning algorithms included in for example +Scikit-Learn.

+

These examples will serve us the purpose of getting +started. Furthermore, they allow us to catch more than two birds with +a stone. They will allow us to bring in some programming specific +topics and tools as well as showing the power of various Python +libraries for machine learning and statistical data analysis.

+

Here, we will mainly focus on two +specific Python packages for Machine Learning, Scikit-Learn and +Tensorflow (see below for links etc). Moreover, the examples we +introduce will serve as inputs to many of our discussions later, as +well as allowing you to set up models and produce your own data and +get started with programming.

+
+
+

1.2. What is Machine Learning?

+

Statistics, data science and machine learning form important fields of +research in modern science. They describe how to learn and make +predictions from data, as well as allowing us to extract important +correlations about physical process and the underlying laws of motion +in large data sets. The latter, big data sets, appear frequently in +essentially all disciplines, from the traditional Science, Technology, +Mathematics and Engineering fields to Life Science, Law, education +research, the Humanities and the Social Sciences.

+

It has become more +and more common to see research projects on big data in for example +the Social Sciences where extracting patterns from complicated survey +data is one of many research directions. Having a solid grasp of data +analysis and machine learning is thus becoming central to scientific +computing in many fields, and competences and skills within the fields +of machine learning and scientific computing are nowadays strongly +requested by many potential employers. The latter cannot be +overstated, familiarity with machine learning has almost become a +prerequisite for many of the most exciting employment opportunities, +whether they are in bioinformatics, life science, physics or finance, +in the private or the public sector. This author has had several +students or met students who have been hired recently based on their +skills and competences in scientific computing and data science, often +with marginal knowledge of machine learning.

+

Machine learning is a subfield of computer science, and is closely +related to computational statistics. It evolved from the study of +pattern recognition in artificial intelligence (AI) research, and has +made contributions to AI tasks like computer vision, natural language +processing and speech recognition. Many of the methods we will study are also +strongly rooted in basic mathematics and physics research.

+

Ideally, machine learning represents the science of giving computers +the ability to learn without being explicitly programmed. The idea is +that there exist generic algorithms which can be used to find patterns +in a broad class of data sets without having to write code +specifically for each problem. The algorithm will build its own logic +based on the data. You should however always keep in mind that +machines and algorithms are to a large extent developed by humans. The +insights and knowledge we have about a specific system, play a central +role when we develop a specific machine learning algorithm.

+

Machine learning is an extremely rich field, in spite of its young +age. The increases we have seen during the last three decades in +computational capabilities have been followed by developments of +methods and techniques for analyzing and handling large date sets, +relying heavily on statistics, computer science and mathematics. The +field is rather new and developing rapidly. Popular software packages +written in Python for machine learning like +Scikit-learn, +Tensorflow, +PyTorch and Keras, all +freely available at their respective GitHub sites, encompass +communities of developers in the thousands or more. And the number of +code developers and contributors keeps increasing. Not all the +algorithms and methods can be given a rigorous mathematical +justification, opening up thereby large rooms for experimenting and +trial and error and thereby exciting new developments. However, a +solid command of linear algebra, multivariate theory, probability +theory, statistical data analysis, understanding errors and Monte +Carlo methods are central elements in a proper understanding of many +of algorithms and methods we will discuss.

+

The approaches to machine learning are many, but are often split into +two main categories. In supervised learning we know the answer to a +problem, and let the computer deduce the logic behind it. On the other +hand, unsupervised learning is a method for finding patterns and +relationship in data sets without any prior knowledge of the system. +Some authours also operate with a third category, namely +reinforcement learning. This is a paradigm of learning inspired by +behavioral psychology, where learning is achieved by trial-and-error, +solely from rewards and punishment.

+

Another way to categorize machine learning tasks is to consider the +desired output of a system. Some of the most common tasks are:

+
    +
  • Classification: Outputs are divided into two or more classes. The goal is to produce a model that assigns inputs into one of these classes. An example is to identify digits based on pictures of hand-written ones. Classification is typically supervised learning.

  • +
  • Regression: Finding a functional relationship between an input data set and a reference data set. The goal is to construct a function that maps input data to continuous output values.

  • +
  • Clustering: Data are divided into groups with certain common traits, without knowing the different groups beforehand. It is thus a form of unsupervised learning.

  • +
+

The methods we cover have three main topics in common, irrespective of +whether we deal with supervised or unsupervised learning. The first +ingredient is normally our data set (which can be subdivided into +training and test data), the second item is a model which is normally a +function of some parameters. The model reflects our knowledge of the system (or lack thereof). As an example, if we know that our data show a behavior similar to what would be predicted by a polynomial, fitting our data to a polynomial of some degree would then determin our model.

+

The last ingredient is a so-called cost +function which allows us to present an estimate on how good our model +is in reproducing the data it is supposed to train.
+At the heart of basically all ML algorithms there are so-called minimization algorithms, often we end up with various variants of gradient methods.

+
+
+

1.3. Software and needed installations

+

We will make extensive use of Python as programming language and its +myriad of available libraries. You will find +Jupyter notebooks invaluable in your work. You can run R +codes in the Jupyter/IPython notebooks, with the immediate benefit of +visualizing your data. You can also use compiled languages like C++, +Rust, Julia, Fortran etc if you prefer. The focus in these lectures will be +on Python.

+

If you have Python installed (we strongly recommend Python3) and you feel +pretty familiar with installing different packages, we recommend that +you install the following Python packages via pip as

+
    +
  1. pip install numpy scipy matplotlib ipython scikit-learn mglearn sympy pandas pillow

  2. +
+

For Python3, replace pip with pip3.

+

For OSX users we recommend, after having installed Xcode, to +install brew. Brew allows for a seamless installation of additional +software via for example

+
    +
  1. brew install python3

  2. +
+

For Linux users, with its variety of distributions like for example the widely popular Ubuntu distribution, +you can use pip as well and simply install Python as

+
    +
  1. sudo apt-get install python3 (or python for pyhton2.7)

  2. +
+

etc etc.

+
+
+

1.4. Python installers

+

If you don’t want to perform these operations separately and venture +into the hassle of exploring how to set up dependencies and paths, we +recommend two widely used distrubutions which set up all relevant +dependencies for Python, namely

+ +

which is an open source +distribution of the Python and R programming languages for large-scale +data processing, predictive analytics, and scientific computing, that +aims to simplify package management and deployment. Package versions +are managed by the package management system conda.

+ +

is a Python +distribution for scientific and analytic computing distribution and +analysis environment, available for free and under a commercial +license.

+

Furthermore, Google’s Colab is a free Jupyter notebook environment that requires +no setup and runs entirely in the cloud. Try it out!

+
+
+

1.5. Useful Python libraries

+

Here we list several useful Python libraries we strongly recommend (if you use anaconda many of these are already there)

+
    +
  • NumPy is a highly popular library for large, multi-dimensional arrays and matrices, along with a large collection of high-level mathematical functions to operate on these arrays

  • +
  • The pandas library provides high-performance, easy-to-use data structures and data analysis tools

  • +
  • Xarray is a Python package that makes working with labelled multi-dimensional arrays simple, efficient, and fun!

  • +
  • Scipy (pronounced “Sigh Pie”) is a Python-based ecosystem of open-source software for mathematics, science, and engineering.

  • +
  • Matplotlib is a Python 2D plotting library which produces publication quality figures in a variety of hardcopy formats and interactive environments across platforms.

  • +
  • Autograd can automatically differentiate native Python and Numpy code. It can handle a large subset of Python’s features, including loops, ifs, recursion and closures, and it can even take derivatives of derivatives of derivatives

  • +
  • SymPy is a Python library for symbolic mathematics.

  • +
  • scikit-learn has simple and efficient tools for machine learning, data mining and data analysis

  • +
  • TensorFlow is a Python library for fast numerical computing created and released by Google

  • +
  • Keras is a high-level neural networks API, written in Python and capable of running on top of TensorFlow, CNTK, or Theano

  • +
  • And many more such as pytorch, Theano etc

  • +
+
+
+

1.6. Installing R, C++, cython or Julia

+

You will also find it convenient to utilize R. We will mainly +use Python during our lectures and in various projects and exercises. +Those of you +already familiar with R should feel free to continue using R, keeping +however an eye on the parallel Python set ups. Similarly, if you are a +Python afecionado, feel free to explore R as well. Jupyter/Ipython +notebook allows you to run R codes interactively in your +browser. The software library R is really tailored for statistical data analysis +and allows for an easy usage of the tools and algorithms we will discuss in these +lectures.

+

To install R with Jupyter notebook +follow the link here

+
+
+

1.7. Installing R, C++, cython, Numba etc

+

For the C++ aficionados, Jupyter/IPython notebook allows you also to +install C++ and run codes written in this language interactively in +the browser. Since we will emphasize writing many of the algorithms +yourself, you can thus opt for either Python or C++ (or Fortran or other compiled languages) as programming +languages.

+

To add more entropy, cython can also be used when running your +notebooks. It means that Python with the jupyter notebook +setup allows you to integrate widely popular softwares and tools for +scientific computing. Similarly, the +Numba Python package delivers increased performance +capabilities with minimal rewrites of your codes. With its +versatility, including symbolic operations, Python offers a unique +computational environment. Your jupyter notebook can easily be +converted into a nicely rendered PDF file or a Latex file for +further processing. For example, convert to latex as

+
    pycod jupyter nbconvert filename.ipynb --to latex 
+
+
+

And to add more versatility, the Python package SymPy is a Python library for symbolic mathematics. It aims to become a full-featured computer algebra system (CAS) and is entirely written in Python.

+

Finally, if you wish to use the light mark-up language +doconce you can convert a standard ascii text file into various HTML +formats, ipython notebooks, latex files, pdf files etc with minimal edits. These lectures were generated using doconce.

+
+
+

1.8. Numpy examples and Important Matrix and vector handling packages

+

There are several central software libraries for linear algebra and eigenvalue problems. Several of the more +popular ones have been wrapped into ofter software packages like those from the widely used text Numerical Recipes. The original source codes in many of the available packages are often taken from the widely used +software package LAPACK, which follows two other popular packages +developed in the 1970s, namely EISPACK and LINPACK. We describe them shortly here.

+
    +
  • LINPACK: package for linear equations and least square problems.

  • +
  • LAPACK:package for solving symmetric, unsymmetric and generalized eigenvalue problems. From LAPACK’s website http://www.netlib.org it is possible to download for free all source codes from this library. Both C/C++ and Fortran versions are available.

  • +
  • BLAS (I, II and III): (Basic Linear Algebra Subprograms) are routines that provide standard building blocks for performing basic vector and matrix operations. Blas I is vector operations, II vector-matrix operations and III matrix-matrix operations. Highly parallelized and efficient codes, all available for download from http://www.netlib.org.

  • +
+
+
+

1.9. Basic Matrix Features

+

Matrix properties reminder

+
+\[\begin{split} +\mathbf{A} = + \begin{bmatrix} a_{11} & a_{12} & a_{13} & a_{14} \\ + a_{21} & a_{22} & a_{23} & a_{24} \\ + a_{31} & a_{32} & a_{33} & a_{34} \\ + a_{41} & a_{42} & a_{43} & a_{44} + \end{bmatrix}\qquad +\mathbf{I} = + \begin{bmatrix} 1 & 0 & 0 & 0 \\ + 0 & 1 & 0 & 0 \\ + 0 & 0 & 1 & 0 \\ + 0 & 0 & 0 & 1 + \end{bmatrix} +\end{split}\]
+

The inverse of a matrix is defined by

+
+\[ +\mathbf{A}^{-1} \cdot \mathbf{A} = I +\]
+ + + + + + + + + + + +
Relations Name matrix elements
$A = A^{T}$ symmetric $a_{ij} = a_{ji}$
$A = \left (A^{T} \right )^{-1}$ real orthogonal $\sum_k a_{ik} a_{jk} = \sum_k a_{ki} a_{kj} = \delta_{ij}$
$A = A^{ * }$ real matrix $a_{ij} = a_{ij}^{ * }$
$A = A^{\dagger}$ hermitian $a_{ij} = a_{ji}^{ * }$
$A = \left (A^{\dagger} \right )^{-1}$ unitary $\sum_k a_{ik} a_{jk}^{ * } = \sum_k a_{ki}^{ * } a_{kj} = \delta_{ij}$
+
+

1.9.1. Some famous Matrices

+
    +
  • Diagonal if \(a_{ij}=0\) for \(i\ne j\)

  • +
  • Upper triangular if \(a_{ij}=0\) for \(i > j\)

  • +
  • Lower triangular if \(a_{ij}=0\) for \(i < j\)

  • +
  • Upper Hessenberg if \(a_{ij}=0\) for \(i > j+1\)

  • +
  • Lower Hessenberg if \(a_{ij}=0\) for \(i < j+1\)

  • +
  • Tridiagonal if \(a_{ij}=0\) for \(|i -j| > 1\)

  • +
  • Lower banded with bandwidth \(p\): \(a_{ij}=0\) for \(i > j+p\)

  • +
  • Upper banded with bandwidth \(p\): \(a_{ij}=0\) for \(i < j+p\)

  • +
  • Banded, block upper triangular, block lower triangular….

  • +
+
+
+

1.9.2. More Basic Matrix Features

+

Some Equivalent Statements +For an \(N\times N\) matrix \(\mathbf{A}\) the following properties are all equivalent

+
    +
  • If the inverse of \(\mathbf{A}\) exists, \(\mathbf{A}\) is nonsingular.

  • +
  • The equation \(\mathbf{Ax}=0\) implies \(\mathbf{x}=0\).

  • +
  • The rows of \(\mathbf{A}\) form a basis of \(R^N\).

  • +
  • The columns of \(\mathbf{A}\) form a basis of \(R^N\).

  • +
  • \(\mathbf{A}\) is a product of elementary matrices.

  • +
  • \(0\) is not eigenvalue of \(\mathbf{A}\).

  • +
+
+
+
+

1.10. Numpy and arrays

+

Numpy provides an easy way to handle arrays in Python. The standard way to import this library is as

+
+
+
import numpy as np
+
+
+
+
+

Here follows a simple example where we set up an array of ten elements, all determined by random numbers drawn according to the normal distribution,

+
+
+
n = 10
+x = np.random.normal(size=n)
+print(x)
+
+
+
+
+
[ 1.01043351  0.90745348  0.55248701 -0.3588324   0.26881845  0.56326636
+  0.10659197 -0.97644907  0.16328353 -2.30075596]
+
+
+
+
+

We defined a vector \(x\) with \(n=10\) elements with its values given by the Normal distribution \(N(0,1)\). +Another alternative is to declare a vector as follows

+
+
+
import numpy as np
+x = np.array([1, 2, 3])
+print(x)
+
+
+
+
+
[1 2 3]
+
+
+
+
+

Here we have defined a vector with three elements, with \(x_0=1\), \(x_1=2\) and \(x_2=3\). Note that both Python and C++ +start numbering array elements from \(0\) and on. This means that a vector with \(n\) elements has a sequence of entities \(x_0, x_1, x_2, \dots, x_{n-1}\). We could also let (recommended) Numpy to compute the logarithms of a specific array as

+
+
+
import numpy as np
+x = np.log(np.array([4, 7, 8]))
+print(x)
+
+
+
+
+
[1.38629436 1.94591015 2.07944154]
+
+
+
+
+

In the last example we used Numpy’s unary function \(np.log\). This function is +highly tuned to compute array elements since the code is vectorized +and does not require looping. We normaly recommend that you use the +Numpy intrinsic functions instead of the corresponding log function +from Python’s math module. The looping is done explicitely by the +np.log function. The alternative, and slower way to compute the +logarithms of a vector would be to write

+
+
+
import numpy as np
+from math import log
+x = np.array([4, 7, 8])
+for i in range(0, len(x)):
+    x[i] = log(x[i])
+print(x)
+
+
+
+
+
[1 1 2]
+
+
+
+
+

We note that our code is much longer already and we need to import the log function from the math module. +The attentive reader will also notice that the output is \([1, 1, 2]\). Python interprets automagically our numbers as integers (like the automatic keyword in C++). To change this we could define our array elements to be double precision numbers as

+
+
+
import numpy as np
+x = np.log(np.array([4, 7, 8], dtype = np.float64))
+print(x)
+
+
+
+
+
[1.38629436 1.94591015 2.07944154]
+
+
+
+
+

or simply write them as double precision numbers (Python uses 64 bits as default for floating point type variables), that is

+
+
+
import numpy as np
+x = np.log(np.array([4.0, 7.0, 8.0])
+print(x)
+
+
+
+
+
  File "<ipython-input-7-f6d7a289d493>", line 3
+    print(x)
+    ^
+SyntaxError: invalid syntax
+
+
+
+
+

To check the number of bytes (remember that one byte contains eight bits for double precision variables), you can use simple use the itemsize functionality (the array \(x\) is actually an object which inherits the functionalities defined in Numpy) as

+
+
+
import numpy as np
+x = np.log(np.array([4.0, 7.0, 8.0])
+print(x.itemsize)
+
+
+
+
+
+
+

1.11. Matrices in Python

+

Having defined vectors, we are now ready to try out matrices. We can +define a \(3 \times 3 \) real matrix \(\hat{A}\) as (recall that we user +lowercase letters for vectors and uppercase letters for matrices)

+
+
+
import numpy as np
+A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))
+print(A)
+
+
+
+
+

If we use the shape function we would get \((3, 3)\) as output, that is verifying that our matrix is a \(3\times 3\) matrix. We can slice the matrix and print for example the first column (Python organized matrix elements in a row-major order, see below) as

+
+
+
import numpy as np
+A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))
+# print the first column, row-major order and elements start with 0
+print(A[:,0])
+
+
+
+
+

We can continue this was by printing out other columns or rows. The example here prints out the second column

+
+
+
import numpy as np
+A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))
+# print the first column, row-major order and elements start with 0
+print(A[1,:])
+
+
+
+
+

Numpy contains many other functionalities that allow us to slice, subdivide etc etc arrays. We strongly recommend that you look up the Numpy website for more details. Useful functions when defining a matrix are the np.zeros function which declares a matrix of a given dimension and sets all elements to zero

+
+
+
import numpy as np
+n = 10
+# define a matrix of dimension 10 x 10 and set all elements to zero
+A = np.zeros( (n, n) )
+print(A)
+
+
+
+
+

or initializing all elements to

+
+
+
import numpy as np
+n = 10
+# define a matrix of dimension 10 x 10 and set all elements to one
+A = np.ones( (n, n) )
+print(A)
+
+
+
+
+

or as unitarily distributed random numbers (see the material on random number generators in the statistics part)

+
+
+
import numpy as np
+n = 10
+# define a matrix of dimension 10 x 10 and set all elements to random numbers with x \in [0, 1]
+A = np.random.rand(n, n)
+print(A)
+
+
+
+
+

As we will see throughout these lectures, there are several extremely useful functionalities in Numpy. +As an example, consider the discussion of the covariance matrix. Suppose we have defined three vectors +\(\hat{x}, \hat{y}, \hat{z}\) with \(n\) elements each. The covariance matrix is defined as

+
+\[\begin{split} +\hat{\Sigma} = \begin{bmatrix} \sigma_{xx} & \sigma_{xy} & \sigma_{xz} \\ + \sigma_{yx} & \sigma_{yy} & \sigma_{yz} \\ + \sigma_{zx} & \sigma_{zy} & \sigma_{zz} + \end{bmatrix}, +\end{split}\]
+

where for example

+
+\[ +\sigma_{xy} =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +\]
+

The Numpy function np.cov calculates the covariance elements using the factor \(1/(n-1)\) instead of \(1/n\) since it assumes we do not have the exact mean values. +The following simple function uses the np.vstack function which takes each vector of dimension \(1\times n\) and produces a \(3\times n\) matrix \(\hat{W}\)

+
+\[\begin{split} +\hat{W} = \begin{bmatrix} x_0 & y_0 & z_0 \\ + x_1 & y_1 & z_1 \\ + x_2 & y_2 & z_2 \\ + \dots & \dots & \dots \\ + x_{n-2} & y_{n-2} & z_{n-2} \\ + x_{n-1} & y_{n-1} & z_{n-1} + \end{bmatrix}, +\end{split}\]
+

which in turn is converted into into the \(3\times 3\) covariance matrix +\(\hat{\Sigma}\) via the Numpy function np.cov(). We note that we can also calculate +the mean value of each set of samples \(\hat{x}\) etc using the Numpy +function np.mean(x). We can also extract the eigenvalues of the +covariance matrix through the np.linalg.eig() function.

+
+
+
# Importing various packages
+import numpy as np
+
+n = 100
+x = np.random.normal(size=n)
+print(np.mean(x))
+y = 4+3*x+np.random.normal(size=n)
+print(np.mean(y))
+z = x**3+np.random.normal(size=n)
+print(np.mean(z))
+W = np.vstack((x, y, z))
+Sigma = np.cov(W)
+print(Sigma)
+Eigvals, Eigvecs = np.linalg.eig(Sigma)
+print(Eigvals)
+
+
+
+
+
+
+
%matplotlib inline
+
+import numpy as np
+import matplotlib.pyplot as plt
+from scipy import sparse
+eye = np.eye(4)
+print(eye)
+sparse_mtx = sparse.csr_matrix(eye)
+print(sparse_mtx)
+x = np.linspace(-10,10,100)
+y = np.sin(x)
+plt.plot(x,y,marker='x')
+plt.show()
+
+
+
+
+
+
+

1.12. Meet the Pandas

+ +

Another useful Python package is +pandas, which is an open source library +providing high-performance, easy-to-use data structures and data +analysis tools for Python. pandas stands for panel data, a term borrowed from econometrics and is an efficient library for data analysis with an emphasis on tabular data. +pandas has two major classes, the DataFrame class with two-dimensional data objects and tabular data organized in columns and the class Series with a focus on one-dimensional data objects. Both classes allow you to index data easily as we will see in the examples below. +pandas allows you also to perform mathematical operations on the data, spanning from simple reshapings of vectors and matrices to statistical operations.

+

The following simple example shows how we can, in an easy way make tables of our data. Here we define a data set which includes names, place of birth and date of birth, and displays the data in an easy to read way. We will see repeated use of pandas, in particular in connection with classification of data.

+
+
+
import pandas as pd
+from IPython.display import display
+data = {'First Name': ["Frodo", "Bilbo", "Aragorn II", "Samwise"],
+        'Last Name': ["Baggins", "Baggins","Elessar","Gamgee"],
+        'Place of birth': ["Shire", "Shire", "Eriador", "Shire"],
+        'Date of Birth T.A.': [2968, 2890, 2931, 2980]
+        }
+data_pandas = pd.DataFrame(data)
+display(data_pandas)
+
+
+
+
+

In the above we have imported pandas with the shorthand pd, the latter has become the standard way we import pandas. We make then a list of various variables +and reorganize the aboves lists into a DataFrame and then print out a neat table with specific column labels as Name, place of birth and date of birth. +Displaying these results, we see that the indices are given by the default numbers from zero to three. +pandas is extremely flexible and we can easily change the above indices by defining a new type of indexing as

+
+
+
data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam'])
+display(data_pandas)
+
+
+
+
+

Thereafter we display the content of the row which begins with the index Aragorn

+
+
+
display(data_pandas.loc['Aragorn'])
+
+
+
+
+

We can easily append data to this, for example

+
+
+
new_hobbit = {'First Name': ["Peregrin"],
+              'Last Name': ["Took"],
+              'Place of birth': ["Shire"],
+              'Date of Birth T.A.': [2990]
+              }
+data_pandas=data_pandas.append(pd.DataFrame(new_hobbit, index=['Pippin']))
+display(data_pandas)
+
+
+
+
+

Here are other examples where we use the DataFrame functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix +of dimensionality \(10\times 5\) and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations.

+
+
+
import numpy as np
+import pandas as pd
+from IPython.display import display
+np.random.seed(100)
+# setting up a 10 x 5 matrix
+rows = 10
+cols = 5
+a = np.random.randn(rows,cols)
+df = pd.DataFrame(a)
+display(df)
+print(df.mean())
+print(df.std())
+display(df**2)
+
+
+
+
+

Thereafter we can select specific columns only and plot final results

+
+
+
df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']
+df.index = np.arange(10)
+
+display(df)
+print(df['Second'].mean() )
+print(df.info())
+print(df.describe())
+
+from pylab import plt, mpl
+plt.style.use('seaborn')
+mpl.rcParams['font.family'] = 'serif'
+
+df.cumsum().plot(lw=2.0, figsize=(10,6))
+plt.show()
+
+
+df.plot.bar(figsize=(10,6), rot=15)
+plt.show()
+
+
+
+
+

We can produce a \(4\times 4\) matrix

+
+
+
b = np.arange(16).reshape((4,4))
+print(b)
+df1 = pd.DataFrame(b)
+print(df1)
+
+
+
+
+

and many other operations.

+

The Series class is another important class included in +pandas. You can view it as a specialization of DataFrame but where +we have just a single column of data. It shares many of the same features as _DataFrame. As with DataFrame, +most operations are vectorized, achieving thereby a high performance when dealing with computations of arrays, in particular labeled arrays. +As we will see below it leads also to a very concice code close to the mathematical operations we may be interested in. +For multidimensional arrays, we recommend strongly xarray. xarray has much of the same flexibility as pandas, but allows for the extension to higher dimensions than two. We will see examples later of the usage of both pandas and xarray.

+

In order to study various Machine Learning algorithms, we need to +access data. Acccessing data is an essential step in all machine +learning algorithms. In particular, setting up the so-called design +matrix (to be defined below) is often the first element we need in +order to perform our calculations. To set up the design matrix means +reading (and later, when the calculations are done, writing) data +in various formats, The formats span from reading files from disk, +loading data from databases and interacting with online sources +like web application programming interfaces (APIs).

+

In handling various input formats, as discussed above, we will mainly stay with pandas, +a Python package which allows us, in a seamless and painless way, to +deal with a multitude of formats, from standard csv (comma separated +values) files, via excel, html to hdf5 formats. With pandas +and the DataFrame and Series functionalities we are able to convert text data +into the calculational formats we need for a specific algorithm. And our code is going to be +pretty close the basic mathematical expressions.

+

Our first data set is going to be a classic from nuclear physics, namely all +available data on binding energies. Don’t be intimidated if you are not familiar with nuclear physics. It serves simply as an example here of a data set.

+

We will show some of the +strengths of packages like Scikit-Learn in fitting nuclear binding energies to +specific functions using linear regression first. Then, as a teaser, we will show you how +you can easily implement other algorithms like decision trees and random forests and neural networks.

+

But before we really start with nuclear physics data, let’s just look at some simpler polynomial fitting cases, such as, +(don’t be offended) fitting straight lines!

+
+
+

1.13. Simple linear regression model using scikit-learn

+

We start with perhaps our simplest possible example, using Scikit-Learn to perform linear regression analysis on a data set produced by us.

+

What follows is a simple Python code where we have defined a function +\(y\) in terms of the variable \(x\). Both are defined as vectors with \(100\) entries. +The numbers in the vector \(\hat{x}\) are given +by random numbers generated with a uniform distribution with entries +\(x_i \in [0,1]\) (more about probability distribution functions +later). These values are then used to define a function \(y(x)\) +(tabulated again as a vector) with a linear dependence on \(x\) plus a +random noise added via the normal distribution.

+

The Numpy functions are imported used the import numpy as np +statement and the random number generator for the uniform distribution +is called using the function np.random.rand(), where we specificy +that we want \(100\) random variables. Using Numpy we define +automatically an array with the specified number of elements, \(100\) in +our case. With the Numpy function randn() we can compute random +numbers with the normal distribution (mean value \(\mu\) equal to zero and +variance \(\sigma^2\) set to one) and produce the values of \(y\) assuming a linear +dependence as function of \(x\)

+
+\[ +y = 2x+N(0,1), +\]
+

where \(N(0,1)\) represents random numbers generated by the normal +distribution. From Scikit-Learn we import then the +LinearRegression functionality and make a prediction \(\tilde{y} = +\alpha + \beta x\) using the function fit(x,y). We call the set of +data \((\hat{x},\hat{y})\) for our training data. The Python package +scikit-learn has also a functionality which extracts the above +fitting parameters \(\alpha\) and \(\beta\) (see below). Later we will +distinguish between training data and test data.

+

For plotting we use the Python package +matplotlib which produces publication +quality figures. Feel free to explore the extensive +gallery of examples. In +this example we plot our original values of \(x\) and \(y\) as well as the +prediction ypredict (\(\tilde{y}\)), which attempts at fitting our +data with a straight line.

+

The Python code follows here.

+
+
+
# Importing various packages
+import numpy as np
+import matplotlib.pyplot as plt
+from sklearn.linear_model import LinearRegression
+
+x = np.random.rand(100,1)
+y = 2*x+np.random.randn(100,1)
+linreg = LinearRegression()
+linreg.fit(x,y)
+xnew = np.array([[0],[1]])
+ypredict = linreg.predict(xnew)
+
+plt.plot(xnew, ypredict, "r-")
+plt.plot(x, y ,'ro')
+plt.axis([0,1.0,0, 5.0])
+plt.xlabel(r'$x$')
+plt.ylabel(r'$y$')
+plt.title(r'Simple Linear Regression')
+plt.show()
+
+
+
+
+

This example serves several aims. It allows us to demonstrate several +aspects of data analysis and later machine learning algorithms. The +immediate visualization shows that our linear fit is not +impressive. It goes through the data points, but there are many +outliers which are not reproduced by our linear regression. We could +now play around with this small program and change for example the +factor in front of \(x\) and the normal distribution. Try to change the +function \(y\) to

+
+\[ +y = 10x+0.01 \times N(0,1), +\]
+

where \(x\) is defined as before. Does the fit look better? Indeed, by +reducing the role of the noise given by the normal distribution we see immediately that +our linear prediction seemingly reproduces better the training +set. However, this testing ‘by the eye’ is obviouly not satisfactory in the +long run. Here we have only defined the training data and our model, and +have not discussed a more rigorous approach to the cost function.

+

We need more rigorous criteria in defining whether we have succeeded or +not in modeling our training data. You will be surprised to see that +many scientists seldomly venture beyond this ‘by the eye’ approach. A +standard approach for the cost function is the so-called \(\chi^2\) +function (a variant of the mean-squared error (MSE))

+
+\[ +\chi^2 = \frac{1}{n} +\sum_{i=0}^{n-1}\frac{(y_i-\tilde{y}_i)^2}{\sigma_i^2}, +\]
+

where \(\sigma_i^2\) is the variance (to be defined later) of the entry +\(y_i\). We may not know the explicit value of \(\sigma_i^2\), it serves +however the aim of scaling the equations and make the cost function +dimensionless.

+

Minimizing the cost function is a central aspect of +our discussions to come. Finding its minima as function of the model +parameters (\(\alpha\) and \(\beta\) in our case) will be a recurring +theme in these series of lectures. Essentially all machine learning +algorithms we will discuss center around the minimization of the +chosen cost function. This depends in turn on our specific +model for describing the data, a typical situation in supervised +learning. Automatizing the search for the minima of the cost function is a +central ingredient in all algorithms. Typical methods which are +employed are various variants of gradient methods. These will be +discussed in more detail later. Again, you’ll be surprised to hear that +many practitioners minimize the above function ‘’by the eye’, popularly dubbed as +‘chi by the eye’. That is, change a parameter and see (visually and numerically) that +the \(\chi^2\) function becomes smaller.

+

There are many ways to define the cost function. A simpler approach is to look at the relative difference between the training data and the predicted data, that is we define +the relative error (why would we prefer the MSE instead of the relative error?) as

+
+\[ +\epsilon_{\mathrm{relative}}= \frac{\vert \hat{y} -\hat{\tilde{y}}\vert}{\vert \hat{y}\vert}. +\]
+

The squared cost function results in an arithmetic mean-unbiased +estimator, and the absolute-value cost function results in a +median-unbiased estimator (in the one-dimensional case, and a +geometric median-unbiased estimator for the multi-dimensional +case). The squared cost function has the disadvantage that it has the tendency +to be dominated by outliers.

+

We can modify easily the above Python code and plot the relative error instead

+
+
+
import numpy as np
+import matplotlib.pyplot as plt
+from sklearn.linear_model import LinearRegression
+
+x = np.random.rand(100,1)
+y = 5*x+0.01*np.random.randn(100,1)
+linreg = LinearRegression()
+linreg.fit(x,y)
+ypredict = linreg.predict(x)
+
+plt.plot(x, np.abs(ypredict-y)/abs(y), "ro")
+plt.axis([0,1.0,0.0, 0.5])
+plt.xlabel(r'$x$')
+plt.ylabel(r'$\epsilon_{\mathrm{relative}}$')
+plt.title(r'Relative error')
+plt.show()
+
+
+
+
+

Depending on the parameter in front of the normal distribution, we may +have a small or larger relative error. Try to play around with +different training data sets and study (graphically) the value of the +relative error.

+

As mentioned above, Scikit-Learn has an impressive functionality. +We can for example extract the values of \(\alpha\) and \(\beta\) and +their error estimates, or the variance and standard deviation and many +other properties from the statistical data analysis.

+

Here we show an +example of the functionality of Scikit-Learn.

+
+
+
import numpy as np 
+import matplotlib.pyplot as plt 
+from sklearn.linear_model import LinearRegression 
+from sklearn.metrics import mean_squared_error, r2_score, mean_squared_log_error, mean_absolute_error
+
+x = np.random.rand(100,1)
+y = 2.0+ 5*x+0.5*np.random.randn(100,1)
+linreg = LinearRegression()
+linreg.fit(x,y)
+ypredict = linreg.predict(x)
+print('The intercept alpha: \n', linreg.intercept_)
+print('Coefficient beta : \n', linreg.coef_)
+# The mean squared error                               
+print("Mean squared error: %.2f" % mean_squared_error(y, ypredict))
+# Explained variance score: 1 is perfect prediction                                 
+print('Variance score: %.2f' % r2_score(y, ypredict))
+# Mean squared log error                                                        
+print('Mean squared log error: %.2f' % mean_squared_log_error(y, ypredict) )
+# Mean absolute error                                                           
+print('Mean absolute error: %.2f' % mean_absolute_error(y, ypredict))
+plt.plot(x, ypredict, "r-")
+plt.plot(x, y ,'ro')
+plt.axis([0.0,1.0,1.5, 7.0])
+plt.xlabel(r'$x$')
+plt.ylabel(r'$y$')
+plt.title(r'Linear Regression fit ')
+plt.show()
+
+
+
+
+

The function coef gives us the parameter \(\beta\) of our fit while intercept yields +\(\alpha\). Depending on the constant in front of the normal distribution, we get values near or far from \(alpha =2\) and \(\beta =5\). Try to play around with different parameters in front of the normal distribution. The function meansquarederror gives us the mean square error, a risk metric corresponding to the expected value of the squared (quadratic) error or loss defined as

+
+\[ +MSE(\hat{y},\hat{\tilde{y}}) = \frac{1}{n} +\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2, +\]
+

The smaller the value, the better the fit. Ideally we would like to +have an MSE equal zero. The attentive reader has probably recognized +this function as being similar to the \(\chi^2\) function defined above.

+

The r2score function computes \(R^2\), the coefficient of +determination. It provides a measure of how well future samples are +likely to be predicted by the model. Best possible score is 1.0 and it +can be negative (because the model can be arbitrarily worse). A +constant model that always predicts the expected value of \(\hat{y}\), +disregarding the input features, would get a \(R^2\) score of \(0.0\).

+

If \(\tilde{\hat{y}}_i\) is the predicted value of the \(i-th\) sample and \(y_i\) is the corresponding true value, then the score \(R^2\) is defined as

+
+\[ +R^2(\hat{y}, \tilde{\hat{y}}) = 1 - \frac{\sum_{i=0}^{n - 1} (y_i - \tilde{y}_i)^2}{\sum_{i=0}^{n - 1} (y_i - \bar{y})^2}, +\]
+

where we have defined the mean value of \(\hat{y}\) as

+
+\[ +\bar{y} = \frac{1}{n} \sum_{i=0}^{n - 1} y_i. +\]
+

Another quantity taht we will meet again in our discussions of regression analysis is +the mean absolute error (MAE), a risk metric corresponding to the expected value of the absolute error loss or what we call the \(l1\)-norm loss. In our discussion above we presented the relative error. +The MAE is defined as follows

+
+\[ +\text{MAE}(\hat{y}, \hat{\tilde{y}}) = \frac{1}{n} \sum_{i=0}^{n-1} \left| y_i - \tilde{y}_i \right|. +\]
+

We present the +squared logarithmic (quadratic) error

+
+\[ +\text{MSLE}(\hat{y}, \hat{\tilde{y}}) = \frac{1}{n} \sum_{i=0}^{n - 1} (\log_e (1 + y_i) - \log_e (1 + \tilde{y}_i) )^2, +\]
+

where \(\log_e (x)\) stands for the natural logarithm of \(x\). This error +estimate is best to use when targets having exponential growth, such +as population counts, average sales of a commodity over a span of +years etc.

+

Finally, another cost function is the Huber cost function used in robust regression.

+

The rationale behind this possible cost function is its reduced +sensitivity to outliers in the data set. In our discussions on +dimensionality reduction and normalization of data we will meet other +ways of dealing with outliers.

+

The Huber cost function is defined as

+
+\[\begin{split} +H_{\delta}(a)={\begin{cases}{\frac {1}{2}}{a^{2}}&{\text{for }}|a|\leq \delta ,\\\delta (|a|-{\frac {1}{2}}\delta ),&{\text{otherwise.}}\end{cases}}}. +\end{split}\]
+

Here \(a=\boldsymbol{y} - \boldsymbol{\tilde{y}}\). +We will discuss in more +detail these and other functions in the various lectures. We conclude this part with another example. Instead of +a linear \(x\)-dependence we study now a cubic polynomial and use the polynomial regression analysis tools of scikit-learn.

+
+
+
import matplotlib.pyplot as plt
+import numpy as np
+import random
+from sklearn.linear_model import Ridge
+from sklearn.preprocessing import PolynomialFeatures
+from sklearn.pipeline import make_pipeline
+from sklearn.linear_model import LinearRegression
+
+x=np.linspace(0.02,0.98,200)
+noise = np.asarray(random.sample((range(200)),200))
+y=x**3*noise
+yn=x**3*100
+poly3 = PolynomialFeatures(degree=3)
+X = poly3.fit_transform(x[:,np.newaxis])
+clf3 = LinearRegression()
+clf3.fit(X,y)
+
+Xplot=poly3.fit_transform(x[:,np.newaxis])
+poly3_plot=plt.plot(x, clf3.predict(Xplot), label='Cubic Fit')
+plt.plot(x,yn, color='red', label="True Cubic")
+plt.scatter(x, y, label='Data', color='orange', s=15)
+plt.legend()
+plt.show()
+
+def error(a):
+    for i in y:
+        err=(y-yn)/yn
+    return abs(np.sum(err))/len(err)
+
+print (error(y))
+
+
+
+
+

Let us now dive into nuclear physics and remind ourselves briefly about some basic features about binding +energies. A basic quantity which can be measured for the ground +states of nuclei is the atomic mass \(M(N, Z)\) of the neutral atom with +atomic mass number \(A\) and charge \(Z\). The number of neutrons is \(N\). There are indeed several sophisticated experiments worldwide which allow us to measure this quantity to high precision (parts per million even).

+

Atomic masses are usually tabulated in terms of the mass excess defined by

+
+\[ +\Delta M(N, Z) = M(N, Z) - uA, +\]
+

where \(u\) is the Atomic Mass Unit

+
+\[ +u = M(^{12}\mathrm{C})/12 = 931.4940954(57) \hspace{0.1cm} \mathrm{MeV}/c^2. +\]
+

The nucleon masses are

+
+\[ +m_p = 1.00727646693(9)u, +\]
+

and

+
+\[ +m_n = 939.56536(8)\hspace{0.1cm} \mathrm{MeV}/c^2 = 1.0086649156(6)u. +\]
+

In the 2016 mass evaluation of by W.J.Huang, G.Audi, M.Wang, F.G.Kondev, S.Naimi and X.Xu +there are data on masses and decays of 3437 nuclei.

+

The nuclear binding energy is defined as the energy required to break +up a given nucleus into its constituent parts of \(N\) neutrons and \(Z\) +protons. In terms of the atomic masses \(M(N, Z)\) the binding energy is +defined by

+
+\[ +BE(N, Z) = ZM_H c^2 + Nm_n c^2 - M(N, Z)c^2 , +\]
+

where \(M_H\) is the mass of the hydrogen atom and \(m_n\) is the mass of the neutron. +In terms of the mass excess the binding energy is given by

+
+\[ +BE(N, Z) = Z\Delta_H c^2 + N\Delta_n c^2 -\Delta(N, Z)c^2 , +\]
+

where \(\Delta_H c^2 = 7.2890\) MeV and \(\Delta_n c^2 = 8.0713\) MeV.

+

A popular and physically intuitive model which can be used to parametrize +the experimental binding energies as function of \(A\), is the so-called +liquid drop model. The ansatz is based on the following expression

+
+\[ +BE(N,Z) = a_1A-a_2A^{2/3}-a_3\frac{Z^2}{A^{1/3}}-a_4\frac{(N-Z)^2}{A}, +\]
+

where \(A\) stands for the number of nucleons and the \(a_i\)s are parameters which are determined by a fit +to the experimental data.

+

To arrive at the above expression we have assumed that we can make the following assumptions:

+
    +
  • There is a volume term \(a_1A\) proportional with the number of nucleons (the energy is also an extensive quantity). When an assembly of nucleons of the same size is packed together into the smallest volume, each interior nucleon has a certain number of other nucleons in contact with it. This contribution is proportional to the volume.

  • +
  • There is a surface energy term \(a_2A^{2/3}\). The assumption here is that a nucleon at the surface of a nucleus interacts with fewer other nucleons than one in the interior of the nucleus and hence its binding energy is less. This surface energy term takes that into account and is therefore negative and is proportional to the surface area.

  • +
  • There is a Coulomb energy term \(a_3\frac{Z^2}{A^{1/3}}\). The electric repulsion between each pair of protons in a nucleus yields less binding.

  • +
  • There is an asymmetry term \(a_4\frac{(N-Z)^2}{A}\). This term is associated with the Pauli exclusion principle and reflects the fact that the proton-neutron interaction is more attractive on the average than the neutron-neutron and proton-proton interactions.

  • +
+

We could also add a so-called pairing term, which is a correction term that +arises from the tendency of proton pairs and neutron pairs to +occur. An even number of particles is more stable than an odd number.

+
+

1.13.1. Organizing our data

+

Let us start with reading and organizing our data. +We start with the compilation of masses and binding energies from 2016. +After having downloaded this file to our own computer, we are now ready to read the file and start structuring our data.

+

We start with preparing folders for storing our calculations and the data file over masses and binding energies. We import also various modules that we will find useful in order to present various Machine Learning methods. Here we focus mainly on the functionality of scikit-learn.

+
+
+
# Common imports
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+import sklearn.linear_model as skl
+from sklearn.model_selection import train_test_split
+from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error
+import os
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+    os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+    os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+    os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+    return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+    return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+    plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("MassEval2016.dat"),'r')
+
+
+
+
+

Before we proceed, we define also a function for making our plots. You can obviously avoid this and simply set up various matplotlib commands every time you need them. You may however find it convenient to collect all such commands in one function and simply call this function.

+
+
+
from pylab import plt, mpl
+plt.style.use('seaborn')
+mpl.rcParams['font.family'] = 'serif'
+
+def MakePlot(x,y, styles, labels, axlabels):
+    plt.figure(figsize=(10,6))
+    for i in range(len(x)):
+        plt.plot(x[i], y[i], styles[i], label = labels[i])
+        plt.xlabel(axlabels[0])
+        plt.ylabel(axlabels[1])
+    plt.legend(loc=0)
+
+
+
+
+

Our next step is to read the data on experimental binding energies and +reorganize them as functions of the mass number \(A\), the number of +protons \(Z\) and neutrons \(N\) using pandas. Before we do this it is +always useful (unless you have a binary file or other types of compressed +data) to actually open the file and simply take a look at it!

+

In particular, the program that outputs the final nuclear masses is written in Fortran with a specific format. It means that we need to figure out the format and which columns contain the data we are interested in. Pandas comes with a function that reads formatted output. After having admired the file, we are now ready to start massaging it with pandas. The file begins with some basic format information.

+
+
+
"""                                                                                                                         
+This is taken from the data file of the mass 2016 evaluation.                                                               
+All files are 3436 lines long with 124 character per line.                                                                  
+       Headers are 39 lines long.                                                                                           
+   col 1     :  Fortran character control: 1 = page feed  0 = line feed                                                     
+   format    :  a1,i3,i5,i5,i5,1x,a3,a4,1x,f13.5,f11.5,f11.3,f9.3,1x,a2,f11.3,f9.3,1x,i3,1x,f12.5,f11.5                     
+   These formats are reflected in the pandas widths variable below, see the statement                                       
+   widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),                                                            
+   Pandas has also a variable header, with length 39 in this case.                                                          
+"""
+
+
+
+
+

The data we are interested in are in columns 2, 3, 4 and 11, giving us +the number of neutrons, protons, mass numbers and binding energies, +respectively. We add also for the sake of completeness the element name. The data are in fixed-width formatted lines and we will +covert them into the pandas DataFrame structure.

+
+
+
# Read the experimental data with Pandas
+Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11),
+              names=('N', 'Z', 'A', 'Element', 'Ebinding'),
+              widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),
+              header=39,
+              index_col=False)
+
+# Extrapolated values are indicated by '#' in place of the decimal place, so
+# the Ebinding column won't be numeric. Coerce to float and drop these entries.
+Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce')
+Masses = Masses.dropna()
+# Convert from keV to MeV.
+Masses['Ebinding'] /= 1000
+
+# Group the DataFrame by nucleon number, A.
+Masses = Masses.groupby('A')
+# Find the rows of the grouped DataFrame with the maximum binding energy.
+Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()])
+
+
+
+
+

We have now read in the data, grouped them according to the variables we are interested in. +We see how easy it is to reorganize the data using pandas. If we +were to do these operations in C/C++ or Fortran, we would have had to +write various functions/subroutines which perform the above +reorganizations for us. Having reorganized the data, we can now start +to make some simple fits using both the functionalities in numpy and +Scikit-Learn afterwards.

+

Now we define five variables which contain +the number of nucleons \(A\), the number of protons \(Z\) and the number of neutrons \(N\), the element name and finally the energies themselves.

+
+
+
A = Masses['A']
+Z = Masses['Z']
+N = Masses['N']
+Element = Masses['Element']
+Energies = Masses['Ebinding']
+print(Masses)
+
+
+
+
+

The next step, and we will define this mathematically later, is to set up the so-called design matrix. We will throughout call this matrix \(\boldsymbol{X}\). +It has dimensionality \(p\times n\), where \(n\) is the number of data points and \(p\) are the so-called predictors. In our case here they are given by the number of polynomials in \(A\) we wish to include in the fit.

+
+
+
# Now we set up the design matrix X
+X = np.zeros((len(A),5))
+X[:,0] = 1
+X[:,1] = A
+X[:,2] = A**(2.0/3.0)
+X[:,3] = A**(-1.0/3.0)
+X[:,4] = A**(-1.0)
+
+
+
+
+

With scikitlearn we are now ready to use linear regression and fit our data.

+
+
+
clf = skl.LinearRegression().fit(X, Energies)
+fity = clf.predict(X)
+
+
+
+
+

Pretty simple!
+Now we can print measures of how our fit is doing, the coefficients from the fits and plot the final fit together with our data.

+
+
+
# The mean squared error                               
+print("Mean squared error: %.2f" % mean_squared_error(Energies, fity))
+# Explained variance score: 1 is perfect prediction                                 
+print('Variance score: %.2f' % r2_score(Energies, fity))
+# Mean absolute error                                                           
+print('Mean absolute error: %.2f' % mean_absolute_error(Energies, fity))
+print(clf.coef_, clf.intercept_)
+
+Masses['Eapprox']  = fity
+# Generate a plot comparing the experimental with the fitted values values.
+fig, ax = plt.subplots()
+ax.set_xlabel(r'$A = N + Z$')
+ax.set_ylabel(r'$E_\mathrm{bind}\,/\mathrm{MeV}$')
+ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,
+            label='Ame2016')
+ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',
+            label='Fit')
+ax.legend()
+save_fig("Masses2016")
+plt.show()
+
+
+
+
+

As a teaser, let us now see how we can do this with decision trees using scikit-learn. Later we will switch to so-called random forests!

+
+
+
#Decision Tree Regression
+from sklearn.tree import DecisionTreeRegressor
+regr_1=DecisionTreeRegressor(max_depth=5)
+regr_2=DecisionTreeRegressor(max_depth=7)
+regr_3=DecisionTreeRegressor(max_depth=9)
+regr_1.fit(X, Energies)
+regr_2.fit(X, Energies)
+regr_3.fit(X, Energies)
+
+
+y_1 = regr_1.predict(X)
+y_2 = regr_2.predict(X)
+y_3=regr_3.predict(X)
+Masses['Eapprox'] = y_3
+# Plot the results
+plt.figure()
+plt.plot(A, Energies, color="blue", label="Data", linewidth=2)
+plt.plot(A, y_1, color="red", label="max_depth=5", linewidth=2)
+plt.plot(A, y_2, color="green", label="max_depth=7", linewidth=2)
+plt.plot(A, y_3, color="m", label="max_depth=9", linewidth=2)
+
+plt.xlabel("$A$")
+plt.ylabel("$E$[MeV]")
+plt.title("Decision Tree Regression")
+plt.legend()
+save_fig("Masses2016Trees")
+plt.show()
+print(Masses)
+print(np.mean( (Energies-y_1)**2))
+
+
+
+
+

The seaborn package allows us to visualize data in an efficient way. Note that we use scikit-learn’s multi-layer perceptron (or feed forward neural network) +functionality.

+
+
+
from sklearn.neural_network import MLPRegressor
+from sklearn.metrics import accuracy_score
+import seaborn as sns
+
+X_train = X
+Y_train = Energies
+n_hidden_neurons = 100
+epochs = 100
+# store models for later use
+eta_vals = np.logspace(-5, 1, 7)
+lmbd_vals = np.logspace(-5, 1, 7)
+# store the models for later use
+DNN_scikit = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)
+train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))
+sns.set()
+for i, eta in enumerate(eta_vals):
+    for j, lmbd in enumerate(lmbd_vals):
+        dnn = MLPRegressor(hidden_layer_sizes=(n_hidden_neurons), activation='logistic',
+                            alpha=lmbd, learning_rate_init=eta, max_iter=epochs)
+        dnn.fit(X_train, Y_train)
+        DNN_scikit[i][j] = dnn
+        train_accuracy[i][j] = dnn.score(X_train, Y_train)
+
+fig, ax = plt.subplots(figsize = (10, 10))
+sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis")
+ax.set_title("Training Accuracy")
+ax.set_ylabel("$\eta$")
+ax.set_xlabel("$\lambda$")
+plt.show()
+
+
+
+
+
+
+
+

1.14. Linear Regression, basic elements

+

Video of Lecture.

+

Fitting a continuous function with linear parameterization in terms of the parameters \(\boldsymbol{\beta}\).

+
    +
  • Method of choice for fitting a continuous function!

  • +
  • Gives an excellent introduction to central Machine Learning features with understandable pedagogical links to other methods like Neural Networks, Support Vector Machines etc

  • +
  • Analytical expression for the fitting parameters \(\boldsymbol{\beta}\)

  • +
  • Analytical expressions for statistical propertiers like mean values, variances, confidence intervals and more

  • +
  • Analytical relation with probabilistic interpretations

  • +
  • Easy to introduce basic concepts like bias-variance tradeoff, cross-validation, resampling and regularization techniques and many other ML topics

  • +
  • Easy to code! And links well with classification problems and logistic regression and neural networks

  • +
  • Allows for easy hands-on understanding of gradient descent methods

  • +
  • and many more features

  • +
+

For more discussions of Ridge and Lasso regression, Wessel van Wieringen’s article is highly recommended. +Similarly, Mehta et al’s article is also recommended.

+

Regression modeling deals with the description of the sampling distribution of a given random variable \(y\) and how it varies as function of another variable or a set of such variables \(\boldsymbol{x} =[x_0, x_1,\dots, x_{n-1}]^T\). +The first variable is called the dependent, the outcome or the response variable while the set of variables \(\boldsymbol{x}\) is called the independent variable, or the predictor variable or the explanatory variable.

+

A regression model aims at finding a likelihood function \(p(\boldsymbol{y}\vert \boldsymbol{x})\), that is the conditional distribution for \(\boldsymbol{y}\) with a given \(\boldsymbol{x}\). The estimation of \(p(\boldsymbol{y}\vert \boldsymbol{x})\) is made using a data set with

+
    +
  • \(n\) cases \(i = 0, 1, 2, \dots, n-1\)

  • +
  • Response (target, dependent or outcome) variable \(y_i\) with \(i = 0, 1, 2, \dots, n-1\)

  • +
  • \(p\) so-called explanatory (independent or predictor) variables \(\boldsymbol{x}_i=[x_{i0}, x_{i1}, \dots, x_{ip-1}]\) with \(i = 0, 1, 2, \dots, n-1\) and explanatory variables running from \(0\) to \(p-1\). See below for more explicit examples.

  • +
+

The goal of the regression analysis is to extract/exploit relationship between \(\boldsymbol{y}\) and \(\boldsymbol{x}\) in or to infer causal dependencies, approximations to the likelihood functions, functional relationships and to make predictions, making fits and many other things.

+

Consider an experiment in which \(p\) characteristics of \(n\) samples are +measured. The data from this experiment, for various explanatory variables \(p\) are normally represented by a matrix
+\(\mathbf{X}\).

+

The matrix \(\mathbf{X}\) is called the design +matrix. Additional information of the samples is available in the +form of \(\boldsymbol{y}\) (also as above). The variable \(\boldsymbol{y}\) is +generally referred to as the response variable. The aim of +regression analysis is to explain \(\boldsymbol{y}\) in terms of +\(\boldsymbol{X}\) through a functional relationship like \(y_i = +f(\mathbf{X}_{i,\ast})\). When no prior knowledge on the form of +\(f(\cdot)\) is available, it is common to assume a linear relationship +between \(\boldsymbol{X}\) and \(\boldsymbol{y}\). This assumption gives rise to +the linear regression model where \(\boldsymbol{\beta} = [\beta_0, \ldots, +\beta_{p-1}]^{T}\) are the regression parameters.

+

Linear regression gives us a set of analytical equations for the parameters \(\beta_j\).

+

In order to understand the relation among the predictors \(p\), the set of data \(n\) and the target (outcome, output etc) \(\boldsymbol{y}\), +consider the model we discussed for describing nuclear binding energies.

+

There we assumed that we could parametrize the data using a polynomial approximation based on the liquid drop model. +Assuming

+
+\[ +BE(A) = a_0+a_1A+a_2A^{2/3}+a_3A^{-1/3}+a_4A^{-1}, +\]
+

we have five predictors, that is the intercept, the \(A\) dependent term, the \(A^{2/3}\) term and the \(A^{-1/3}\) and \(A^{-1}\) terms. +This gives \(p=0,1,2,3,4\). Furthermore we have \(n\) entries for each predictor. It means that our design matrix is a +\(p\times n\) matrix \(\boldsymbol{X}\).

+

Here the predictors are based on a model we have made. A popular data set which is widely encountered in ML applications is the +so-called credit card default data from Taiwan. The data set contains data on \(n=30000\) credit card holders with predictors like gender, marital status, age, profession, education, etc. In total there are \(24\) such predictors or attributes leading to a design matrix of dimensionality \(24 \times 30000\). This is however a classification problem and we will come back to it when we discuss Logistic Regression.

+

Before we proceed let us study a case from linear algebra where we aim at fitting a set of data \(\boldsymbol{y}=[y_0,y_1,\dots,y_{n-1}]\). We could think of these data as a result of an experiment or a complicated numerical experiment. These data are functions of a series of variables \(\boldsymbol{x}=[x_0,x_1,\dots,x_{n-1}]\), that is \(y_i = y(x_i)\) with \(i=0,1,2,\dots,n-1\). The variables \(x_i\) could represent physical quantities like time, temperature, position etc. We assume that \(y(x)\) is a smooth function.

+

Since obtaining these data points may not be trivial, we want to use these data to fit a function which can allow us to make predictions for values of \(y\) which are not in the present set. The perhaps simplest approach is to assume we can parametrize our function in terms of a polynomial of degree \(n-1\) with \(n\) points, that is

+
+\[ +y=y(x) \rightarrow y(x_i)=\tilde{y}_i+\epsilon_i=\sum_{j=0}^{n-1} \beta_j x_i^j+\epsilon_i, +\]
+

where \(\epsilon_i\) is the error in our approximation.

+

For every set of values \(y_i,x_i\) we have thus the corresponding set of equations

+
+\[\begin{split} +\begin{align*} +y_0&=\beta_0+\beta_1x_0^1+\beta_2x_0^2+\dots+\beta_{n-1}x_0^{n-1}+\epsilon_0\\ +y_1&=\beta_0+\beta_1x_1^1+\beta_2x_1^2+\dots+\beta_{n-1}x_1^{n-1}+\epsilon_1\\ +y_2&=\beta_0+\beta_1x_2^1+\beta_2x_2^2+\dots+\beta_{n-1}x_2^{n-1}+\epsilon_2\\ +\dots & \dots \\ +y_{n-1}&=\beta_0+\beta_1x_{n-1}^1+\beta_2x_{n-1}^2+\dots+\beta_{n-1}x_{n-1}^{n-1}+\epsilon_{n-1}.\\ +\end{align*} +\end{split}\]
+

Defining the vectors

+
+\[ +\boldsymbol{y} = [y_0,y_1, y_2,\dots, y_{n-1}]^T, +\]
+

and

+
+\[ +\boldsymbol{\beta} = [\beta_0,\beta_1, \beta_2,\dots, \beta_{n-1}]^T, +\]
+

and

+
+\[ +\boldsymbol{\epsilon} = [\epsilon_0,\epsilon_1, \epsilon_2,\dots, \epsilon_{n-1}]^T, +\]
+

and the design matrix

+
+\[\begin{split} +\boldsymbol{X}= +\begin{bmatrix} +1& x_{0}^1 &x_{0}^2& \dots & \dots &x_{0}^{n-1}\\ +1& x_{1}^1 &x_{1}^2& \dots & \dots &x_{1}^{n-1}\\ +1& x_{2}^1 &x_{2}^2& \dots & \dots &x_{2}^{n-1}\\ +\dots& \dots &\dots& \dots & \dots &\dots\\ +1& x_{n-1}^1 &x_{n-1}^2& \dots & \dots &x_{n-1}^{n-1}\\ +\end{bmatrix} +\end{split}\]
+

we can rewrite our equations as

+
+\[ +\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta}+\boldsymbol{\epsilon}. +\]
+

The above design matrix is called a Vandermonde matrix.

+

We are obviously not limited to the above polynomial expansions. We +could replace the various powers of \(x\) with elements of Fourier +series or instead of \(x_i^j\) we could have \(\cos{(j x_i)}\) or \(\sin{(j +x_i)}\), or time series or other orthogonal functions. For every set +of values \(y_i,x_i\) we can then generalize the equations to

+
+\[\begin{split} +\begin{align*} +y_0&=\beta_0x_{00}+\beta_1x_{01}+\beta_2x_{02}+\dots+\beta_{n-1}x_{0n-1}+\epsilon_0\\ +y_1&=\beta_0x_{10}+\beta_1x_{11}+\beta_2x_{12}+\dots+\beta_{n-1}x_{1n-1}+\epsilon_1\\ +y_2&=\beta_0x_{20}+\beta_1x_{21}+\beta_2x_{22}+\dots+\beta_{n-1}x_{2n-1}+\epsilon_2\\ +\dots & \dots \\ +y_{i}&=\beta_0x_{i0}+\beta_1x_{i1}+\beta_2x_{i2}+\dots+\beta_{n-1}x_{in-1}+\epsilon_i\\ +\dots & \dots \\ +y_{n-1}&=\beta_0x_{n-1,0}+\beta_1x_{n-1,2}+\beta_2x_{n-1,2}+\dots+\beta_{n-1}x_{n-1,n-1}+\epsilon_{n-1}.\\ +\end{align*} +\end{split}\]
+

Note that we have \(p=n\) here. The matrix is symmetric. This is generally not the case!

+

We redefine in turn the matrix \(\boldsymbol{X}\) as

+
+\[\begin{split} +\boldsymbol{X}= +\begin{bmatrix} +x_{00}& x_{01} &x_{02}& \dots & \dots &x_{0,n-1}\\ +x_{10}& x_{11} &x_{12}& \dots & \dots &x_{1,n-1}\\ +x_{20}& x_{21} &x_{22}& \dots & \dots &x_{2,n-1}\\ +\dots& \dots &\dots& \dots & \dots &\dots\\ +x_{n-1,0}& x_{n-1,1} &x_{n-1,2}& \dots & \dots &x_{n-1,n-1}\\ +\end{bmatrix} +\end{split}\]
+

and without loss of generality we rewrite again our equations as

+
+\[ +\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta}+\boldsymbol{\epsilon}. +\]
+

The left-hand side of this equation is kwown. Our error vector \(\boldsymbol{\epsilon}\) and the parameter vector \(\boldsymbol{\beta}\) are our unknow quantities. How can we obtain the optimal set of \(\beta_i\) values?

+

We have defined the matrix \(\boldsymbol{X}\) via the equations

+
+\[\begin{split} +\begin{align*} +y_0&=\beta_0x_{00}+\beta_1x_{01}+\beta_2x_{02}+\dots+\beta_{n-1}x_{0n-1}+\epsilon_0\\ +y_1&=\beta_0x_{10}+\beta_1x_{11}+\beta_2x_{12}+\dots+\beta_{n-1}x_{1n-1}+\epsilon_1\\ +y_2&=\beta_0x_{20}+\beta_1x_{21}+\beta_2x_{22}+\dots+\beta_{n-1}x_{2n-1}+\epsilon_1\\ +\dots & \dots \\ +y_{i}&=\beta_0x_{i0}+\beta_1x_{i1}+\beta_2x_{i2}+\dots+\beta_{n-1}x_{in-1}+\epsilon_1\\ +\dots & \dots \\ +y_{n-1}&=\beta_0x_{n-1,0}+\beta_1x_{n-1,2}+\beta_2x_{n-1,2}+\dots+\beta_{n-1}x_{n-1,n-1}+\epsilon_{n-1}.\\ +\end{align*} +\end{split}\]
+

As we noted above, we stayed with a system with the design matrix +\(\boldsymbol{X}\in {\mathbb{R}}^{n\times n}\), that is we have \(p=n\). For reasons to come later (algorithmic arguments) we will hereafter define +our matrix as \(\boldsymbol{X}\in {\mathbb{R}}^{n\times p}\), with the predictors refering to the column numbers and the entries \(n\) being the row elements.

+

In our introductory notes we looked at the so-called liquid drop model. Let us remind ourselves about what we did by looking at the code.

+

We restate the parts of the code we are most interested in.

+
+
+
# Common imports
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from IPython.display import display
+import os
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+    os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+    os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+    os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+    return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+    return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+    plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("MassEval2016.dat"),'r')
+
+
+# Read the experimental data with Pandas
+Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11),
+              names=('N', 'Z', 'A', 'Element', 'Ebinding'),
+              widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),
+              header=39,
+              index_col=False)
+
+# Extrapolated values are indicated by '#' in place of the decimal place, so
+# the Ebinding column won't be numeric. Coerce to float and drop these entries.
+Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce')
+Masses = Masses.dropna()
+# Convert from keV to MeV.
+Masses['Ebinding'] /= 1000
+
+# Group the DataFrame by nucleon number, A.
+Masses = Masses.groupby('A')
+# Find the rows of the grouped DataFrame with the maximum binding energy.
+Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()])
+A = Masses['A']
+Z = Masses['Z']
+N = Masses['N']
+Element = Masses['Element']
+Energies = Masses['Ebinding']
+
+# Now we set up the design matrix X
+X = np.zeros((len(A),5))
+X[:,0] = 1
+X[:,1] = A
+X[:,2] = A**(2.0/3.0)
+X[:,3] = A**(-1.0/3.0)
+X[:,4] = A**(-1.0)
+# Then nice printout using pandas
+DesignMatrix = pd.DataFrame(X)
+DesignMatrix.index = A
+DesignMatrix.columns = ['1', 'A', 'A^(2/3)', 'A^(-1/3)', '1/A']
+display(DesignMatrix)
+
+
+
+
+

With \(\boldsymbol{\beta}\in {\mathbb{R}}^{p\times 1}\), it means that we will hereafter write our equations for the approximation as

+
+\[ +\boldsymbol{\tilde{y}}= \boldsymbol{X}\boldsymbol{\beta}, +\]
+

throughout these lectures.

+

With the above we use the design matrix to define the approximation \(\boldsymbol{\tilde{y}}\) via the unknown quantity \(\boldsymbol{\beta}\) as

+
+\[ +\boldsymbol{\tilde{y}}= \boldsymbol{X}\boldsymbol{\beta}, +\]
+

and in order to find the optimal parameters \(\beta_i\) instead of solving the above linear algebra problem, we define a function which gives a measure of the spread between the values \(y_i\) (which represent hopefully the exact values) and the parameterized values \(\tilde{y}_i\), namely

+
+\[ +C(\boldsymbol{\beta})=\frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)^T\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)\right\}, +\]
+

or using the matrix \(\boldsymbol{X}\) and in a more compact matrix-vector notation as

+
+\[ +C(\boldsymbol{\beta})=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. +\]
+

This function is one possible way to define the so-called cost function.

+

It is also common to define +the function \(C\) as

+
+\[ +C(\boldsymbol{\beta})=\frac{1}{2n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2, +\]
+

since when taking the first derivative with respect to the unknown parameters \(\beta\), the factor of \(2\) cancels out.

+

The function

+
+\[ +C(\boldsymbol{\beta})=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}, +\]
+

can be linked to the variance of the quantity \(y_i\) if we interpret the latter as the mean value. +When linking (see the discussion below) with the maximum likelihood approach below, we will indeed interpret \(y_i\) as a mean value

+
+\[ +y_{i}=\langle y_i \rangle = \beta_0x_{i,0}+\beta_1x_{i,1}+\beta_2x_{i,2}+\dots+\beta_{n-1}x_{i,n-1}+\epsilon_i, +\]
+

where \(\langle y_i \rangle\) is the mean value. Keep in mind also that +till now we have treated \(y_i\) as the exact value. Normally, the +response (dependent or outcome) variable \(y_i\) the outcome of a +numerical experiment or another type of experiment and is thus only an +approximation to the true value. It is then always accompanied by an +error estimate, often limited to a statistical error estimate given by +the standard deviation discussed earlier. In the discussion here we +will treat \(y_i\) as our exact value for the response variable.

+

In order to find the parameters \(\beta_i\) we will then minimize the spread of \(C(\boldsymbol{\beta})\), that is we are going to solve the problem

+
+\[ +{\displaystyle \min_{\boldsymbol{\beta}\in +{\mathbb{R}}^{p}}}\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. +\]
+

In practical terms it means we will require

+
+\[ +\frac{\partial C(\boldsymbol{\beta})}{\partial \beta_j} = \frac{\partial }{\partial \beta_j}\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}\right)^2\right]=0, +\]
+

which results in

+
+\[ +\frac{\partial C(\boldsymbol{\beta})}{\partial \beta_j} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}x_{ij}\left(y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}\right)\right]=0, +\]
+

or in a matrix-vector form as

+
+\[ +\frac{\partial C(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right). +\]
+

We can rewrite

+
+\[ +\frac{\partial C(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right), +\]
+

as

+
+\[ +\boldsymbol{X}^T\boldsymbol{y} = \boldsymbol{X}^T\boldsymbol{X}\boldsymbol{\beta}, +\]
+

and if the matrix \(\boldsymbol{X}^T\boldsymbol{X}\) is invertible we have the solution

+
+\[ +\boldsymbol{\beta} =\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}. +\]
+

We note also that since our design matrix is defined as \(\boldsymbol{X}\in +{\mathbb{R}}^{n\times p}\), the product \(\boldsymbol{X}^T\boldsymbol{X} \in +{\mathbb{R}}^{p\times p}\). In the above case we have that \(p \ll n\), +in our case \(p=5\) meaning that we end up with inverting a small +\(5\times 5\) matrix. This is a rather common situation, in many cases we end up with low-dimensional +matrices to invert. The methods discussed here and for many other +supervised learning algorithms like classification with logistic +regression or support vector machines, exhibit dimensionalities which +allow for the usage of direct linear algebra methods such as LU decomposition or Singular Value Decomposition (SVD) for finding the inverse of the matrix +\(\boldsymbol{X}^T\boldsymbol{X}\).

+

Small question: Do you think the example we have at hand here (the nuclear binding energies) can lead to problems in inverting the matrix \(\boldsymbol{X}^T\boldsymbol{X}\)? What kind of problems can we expect?

+

The following matrix and vector relation will be useful here and for the rest of the course. Vectors are always written as boldfaced lower case letters and +matrices as upper case boldfaced letters.

+

4 +8

+

< +< +< +! +! +M +A +T +H +_ +B +L +O +C +K

+

4 +9

+

< +< +< +! +! +M +A +T +H +_ +B +L +O +C +K

+

5 +0

+

< +< +< +! +! +M +A +T +H +_ +B +L +O +C +K

+
+\[ +\frac{\partial \log{\vert\boldsymbol{A}\vert}}{\partial \boldsymbol{A}} = (\boldsymbol{A}^{-1})^T. +\]
+

The residuals \(\boldsymbol{\epsilon}\) are in turn given by

+
+\[ +\boldsymbol{\epsilon} = \boldsymbol{y}-\boldsymbol{\tilde{y}} = \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}, +\]
+

and with

+
+\[ +\boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)= 0, +\]
+

we have

+
+\[ +\boldsymbol{X}^T\boldsymbol{\epsilon}=\boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)= 0, +\]
+

meaning that the solution for \(\boldsymbol{\beta}\) is the one which minimizes the residuals. Later we will link this with the maximum likelihood approach.

+

Let us now return to our nuclear binding energies and simply code the above equations.

+

It is rather straightforward to implement the matrix inversion and obtain the parameters \(\boldsymbol{\beta}\). After having defined the matrix \(\boldsymbol{X}\) we simply need to +write

+
+
+
# matrix inversion to find beta
+beta = np.linalg.inv(X.T.dot(X)).dot(X.T).dot(Energies)
+# and then make the prediction
+ytilde = X @ beta
+
+
+
+
+

Alternatively, you can use the least squares functionality in Numpy as

+
+
+
fit = np.linalg.lstsq(X, Energies, rcond =None)[0]
+ytildenp = np.dot(fit,X.T)
+
+
+
+
+

And finally we plot our fit with and compare with data

+
+
+
Masses['Eapprox']  = ytilde
+# Generate a plot comparing the experimental with the fitted values values.
+fig, ax = plt.subplots()
+ax.set_xlabel(r'$A = N + Z$')
+ax.set_ylabel(r'$E_\mathrm{bind}\,/\mathrm{MeV}$')
+ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,
+            label='Ame2016')
+ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',
+            label='Fit')
+ax.legend()
+save_fig("Masses2016OLS")
+plt.show()
+
+
+
+
+

We can easily test our fit by computing the \(R2\) score that we discussed in connection with the functionality of Scikit-Learn in the introductory slides. +Since we are not using Scikit-Learn here we can define our own \(R2\) function as

+
+
+
def R2(y_data, y_model):
+    return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
+
+
+
+
+

and we would be using it as

+
+
+
print(R2(Energies,ytilde))
+
+
+
+
+

We can easily add our MSE score as

+
+
+
def MSE(y_data,y_model):
+    n = np.size(y_model)
+    return np.sum((y_data-y_model)**2)/n
+
+print(MSE(Energies,ytilde))
+
+
+
+
+

and finally the relative error as

+
+
+
def RelativeError(y_data,y_model):
+    return abs((y_data-y_model)/y_data)
+print(RelativeError(Energies, ytilde))
+
+
+
+
+
+

1.14.1. The \(\chi^2\) function

+

Normally, the response (dependent or outcome) variable \(y_i\) is the +outcome of a numerical experiment or another type of experiment and is +thus only an approximation to the true value. It is then always +accompanied by an error estimate, often limited to a statistical error +estimate given by the standard deviation discussed earlier. In the +discussion here we will treat \(y_i\) as our exact value for the +response variable.

+

Introducing the standard deviation \(\sigma_i\) for each measurement +\(y_i\), we define now the \(\chi^2\) function (omitting the \(1/n\) term) +as

+
+\[ +\chi^2(\boldsymbol{\beta})=\frac{1}{n}\sum_{i=0}^{n-1}\frac{\left(y_i-\tilde{y}_i\right)^2}{\sigma_i^2}=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)^T\frac{1}{\boldsymbol{\Sigma^2}}\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)\right\}, +\]
+

where the matrix \(\boldsymbol{\Sigma}\) is a diagonal matrix with \(\sigma_i\) as matrix elements.

+

In order to find the parameters \(\beta_i\) we will then minimize the spread of \(\chi^2(\boldsymbol{\beta})\) by requiring

+
+\[ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_j} = \frac{\partial }{\partial \beta_j}\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(\frac{y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}}{\sigma_i}\right)^2\right]=0, +\]
+

which results in

+
+\[ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_j} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}\frac{x_{ij}}{\sigma_i}\left(\frac{y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}}{\sigma_i}\right)\right]=0, +\]
+

or in a matrix-vector form as

+
+\[ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{A}^T\left( \boldsymbol{b}-\boldsymbol{A}\boldsymbol{\beta}\right). +\]
+

where we have defined the matrix \(\boldsymbol{A} =\boldsymbol{X}/\boldsymbol{\Sigma}\) with matrix elements \(a_{ij} = x_{ij}/\sigma_i\) and the vector \(\boldsymbol{b}\) with elements \(b_i = y_i/\sigma_i\).

+

We can rewrite

+
+\[ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{A}^T\left( \boldsymbol{b}-\boldsymbol{A}\boldsymbol{\beta}\right), +\]
+

as

+
+\[ +\boldsymbol{A}^T\boldsymbol{b} = \boldsymbol{A}^T\boldsymbol{A}\boldsymbol{\beta}, +\]
+

and if the matrix \(\boldsymbol{A}^T\boldsymbol{A}\) is invertible we have the solution

+
+\[ +\boldsymbol{\beta} =\left(\boldsymbol{A}^T\boldsymbol{A}\right)^{-1}\boldsymbol{A}^T\boldsymbol{b}. +\]
+

If we then introduce the matrix

+
+\[ +\boldsymbol{H} = \left(\boldsymbol{A}^T\boldsymbol{A}\right)^{-1}, +\]
+

we have then the following expression for the parameters \(\beta_j\) (the matrix elements of \(\boldsymbol{H}\) are \(h_{ij}\))

+
+\[ +\beta_j = \sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}\frac{y_i}{\sigma_i}\frac{x_{ik}}{\sigma_i} = \sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}b_ia_{ik} +\]
+

We state without proof the expression for the uncertainty in the parameters \(\beta_j\) as (we leave this as an exercise)

+
+\[ +\sigma^2(\beta_j) = \sum_{i=0}^{n-1}\sigma_i^2\left( \frac{\partial \beta_j}{\partial y_i}\right)^2, +\]
+

resulting in

+
+\[ +\sigma^2(\beta_j) = \left(\sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}a_{ik}\right)\left(\sum_{l=0}^{p-1}h_{jl}\sum_{m=0}^{n-1}a_{ml}\right) = h_{jj}! +\]
+

The first step here is to approximate the function \(y\) with a first-order polynomial, that is we write

+
+\[ +y=y(x) \rightarrow y(x_i) \approx \beta_0+\beta_1 x_i. +\]
+

By computing the derivatives of \(\chi^2\) with respect to \(\beta_0\) and \(\beta_1\) show that these are given by

+
+\[ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_0} = -2\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(\frac{y_i-\beta_0-\beta_1x_{i}}{\sigma_i^2}\right)\right]=0, +\]
+

and

+
+\[ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_1} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}x_i\left(\frac{y_i-\beta_0-\beta_1x_{i}}{\sigma_i^2}\right)\right]=0. +\]
+

For a linear fit (a first-order polynomial) we don’t need to invert a matrix!!
+Defining

+
+\[ +\gamma = \sum_{i=0}^{n-1}\frac{1}{\sigma_i^2}, +\]
+
+\[ +\gamma_x = \sum_{i=0}^{n-1}\frac{x_{i}}{\sigma_i^2}, +\]
+
+\[ +\gamma_y = \sum_{i=0}^{n-1}\left(\frac{y_i}{\sigma_i^2}\right), +\]
+
+\[ +\gamma_{xx} = \sum_{i=0}^{n-1}\frac{x_ix_{i}}{\sigma_i^2}, +\]
+
+\[ +\gamma_{xy} = \sum_{i=0}^{n-1}\frac{y_ix_{i}}{\sigma_i^2}, +\]
+

we obtain

+
+\[ +\beta_0 = \frac{\gamma_{xx}\gamma_y-\gamma_x\gamma_y}{\gamma\gamma_{xx}-\gamma_x^2}, +\]
+
+\[ +\beta_1 = \frac{\gamma_{xy}\gamma-\gamma_x\gamma_y}{\gamma\gamma_{xx}-\gamma_x^2}. +\]
+

This approach (different linear and non-linear regression) suffers +often from both being underdetermined and overdetermined in the +unknown coefficients \(\beta_i\). A better approach is to use the +Singular Value Decomposition (SVD) method discussed below. Or using +Lasso and Ridge regression. See below.

+
+
+

1.14.2. Fitting an Equation of State for Dense Nuclear Matter

+

Before we continue, let us introduce yet another example. We are going to fit the +nuclear equation of state using results from many-body calculations. +The equation of state we have made available here, as function of +density, has been derived using modern nucleon-nucleon potentials with +the addition of three-body +forces. This +time the file is presented as a standard csv file.

+

The beginning of the Python code here is similar to what you have seen +before, with the same initializations and declarations. We use also +pandas again, rather extensively in order to organize our data.

+

The difference now is that we use Scikit-Learn’s regression tools +instead of our own matrix inversion implementation. Furthermore, we +sneak in Ridge regression (to be discussed below) which includes a +hyperparameter \(\lambda\), also to be explained below.

+
+
+
# Common imports
+import os
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+import matplotlib.pyplot as plt
+import sklearn.linear_model as skl
+from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+    os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+    os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+    os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+    return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+    return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+    plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("EoS.csv"),'r')
+
+# Read the EoS data as  csv file and organize the data into two arrays with density and energies
+EoS = pd.read_csv(infile, names=('Density', 'Energy'))
+EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')
+EoS = EoS.dropna()
+Energies = EoS['Energy']
+Density = EoS['Density']
+#  The design matrix now as function of various polytrops
+X = np.zeros((len(Density),4))
+X[:,3] = Density**(4.0/3.0)
+X[:,2] = Density
+X[:,1] = Density**(2.0/3.0)
+X[:,0] = 1
+
+# We use now Scikit-Learn's linear regressor and ridge regressor
+# OLS part
+clf = skl.LinearRegression().fit(X, Energies)
+ytilde = clf.predict(X)
+EoS['Eols']  = ytilde
+# The mean squared error                               
+print("Mean squared error: %.2f" % mean_squared_error(Energies, ytilde))
+# Explained variance score: 1 is perfect prediction                                 
+print('Variance score: %.2f' % r2_score(Energies, ytilde))
+# Mean absolute error                                                           
+print('Mean absolute error: %.2f' % mean_absolute_error(Energies, ytilde))
+print(clf.coef_, clf.intercept_)
+
+# The Ridge regression with a hyperparameter lambda = 0.1
+_lambda = 0.1
+clf_ridge = skl.Ridge(alpha=_lambda).fit(X, Energies)
+yridge = clf_ridge.predict(X)
+EoS['Eridge']  = yridge
+# The mean squared error                               
+print("Mean squared error: %.2f" % mean_squared_error(Energies, yridge))
+# Explained variance score: 1 is perfect prediction                                 
+print('Variance score: %.2f' % r2_score(Energies, yridge))
+# Mean absolute error                                                           
+print('Mean absolute error: %.2f' % mean_absolute_error(Energies, yridge))
+print(clf_ridge.coef_, clf_ridge.intercept_)
+
+fig, ax = plt.subplots()
+ax.set_xlabel(r'$\rho[\mathrm{fm}^{-3}]$')
+ax.set_ylabel(r'Energy per particle')
+ax.plot(EoS['Density'], EoS['Energy'], alpha=0.7, lw=2,
+            label='Theoretical data')
+ax.plot(EoS['Density'], EoS['Eols'], alpha=0.7, lw=2, c='m',
+            label='OLS')
+ax.plot(EoS['Density'], EoS['Eridge'], alpha=0.7, lw=2, c='g',
+            label='Ridge $\lambda = 0.1$')
+ax.legend()
+save_fig("EoSfitting")
+plt.show()
+
+
+
+
+

The above simple polynomial in density \(\rho\) gives an excellent fit +to the data.

+

We note also that there is a small deviation between the +standard OLS and the Ridge regression at higher densities. We discuss this in more detail +below.

+
+
+
+

1.15. Splitting our Data in Training and Test data

+

It is normal in essentially all Machine Learning studies to split the +data in a training set and a test set (sometimes also an additional +validation set). Scikit-Learn has an own function for this. There +is no explicit recipe for how much data should be included as training +data and say test data. An accepted rule of thumb is to use +approximately \(2/3\) to \(4/5\) of the data as training data. We will +postpone a discussion of this splitting to the end of these notes and +our discussion of the so-called bias-variance tradeoff. Here we +limit ourselves to repeat the above equation of state fitting example +but now splitting the data into a training set and a test set.

+
+
+
import os
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.model_selection import train_test_split
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+    os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+    os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+    os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+    return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+    return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+    plt.savefig(image_path(fig_id) + ".png", format='png')
+
+def R2(y_data, y_model):
+    return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
+def MSE(y_data,y_model):
+    n = np.size(y_model)
+    return np.sum((y_data-y_model)**2)/n
+
+infile = open(data_path("EoS.csv"),'r')
+
+# Read the EoS data as  csv file and organized into two arrays with density and energies
+EoS = pd.read_csv(infile, names=('Density', 'Energy'))
+EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')
+EoS = EoS.dropna()
+Energies = EoS['Energy']
+Density = EoS['Density']
+#  The design matrix now as function of various polytrops
+X = np.zeros((len(Density),5))
+X[:,0] = 1
+X[:,1] = Density**(2.0/3.0)
+X[:,2] = Density
+X[:,3] = Density**(4.0/3.0)
+X[:,4] = Density**(5.0/3.0)
+# We split the data in test and training data
+X_train, X_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2)
+# matrix inversion to find beta
+beta = np.linalg.inv(X_train.T.dot(X_train)).dot(X_train.T).dot(y_train)
+# and then make the prediction
+ytilde = X_train @ beta
+print("Training R2")
+print(R2(y_train,ytilde))
+print("Training MSE")
+print(MSE(y_train,ytilde))
+ypredict = X_test @ beta
+print("Test R2")
+print(R2(y_test,ypredict))
+print("Test MSE")
+print(MSE(y_test,ypredict))
+
+
+
+
+
+
+

1.16. The Boston housing data example

+

The Boston housing
+data set was originally a part of UCI Machine Learning Repository +and has been removed now. The data set is now included in Scikit-Learn’s +library. There are 506 samples and 13 feature (predictor) variables +in this data set. The objective is to predict the value of prices of +the house using the features (predictors) listed here.

+

The features/predictors are

+
    +
  1. CRIM: Per capita crime rate by town

  2. +
  3. ZN: Proportion of residential land zoned for lots over 25000 square feet

  4. +
  5. INDUS: Proportion of non-retail business acres per town

  6. +
  7. CHAS: Charles River dummy variable (= 1 if tract bounds river; 0 otherwise)

  8. +
  9. NOX: Nitric oxide concentration (parts per 10 million)

  10. +
  11. RM: Average number of rooms per dwelling

  12. +
  13. AGE: Proportion of owner-occupied units built prior to 1940

  14. +
  15. DIS: Weighted distances to five Boston employment centers

  16. +
  17. RAD: Index of accessibility to radial highways

  18. +
  19. TAX: Full-value property tax rate per USD10000

  20. +
  21. B: \(1000(Bk - 0.63)^2\), where \(Bk\) is the proportion of [people of African American descent] by town

  22. +
  23. LSTAT: Percentage of lower status of the population

  24. +
  25. MEDV: Median value of owner-occupied homes in USD 1000s

  26. +
+
+
+

1.17. Housing data, the code

+

We start by importing the libraries

+
+
+
import numpy as np
+import matplotlib.pyplot as plt 
+
+import pandas as pd  
+import seaborn as sns
+
+
+
+
+

and load the Boston Housing DataSet from Scikit-Learn

+
+
+
from sklearn.datasets import load_boston
+
+boston_dataset = load_boston()
+
+# boston_dataset is a dictionary
+# let's check what it contains
+boston_dataset.keys()
+
+
+
+
+

Then we invoke Pandas

+
+
+
boston = pd.DataFrame(boston_dataset.data, columns=boston_dataset.feature_names)
+boston.head()
+boston['MEDV'] = boston_dataset.target
+
+
+
+
+

and preprocess the data

+
+
+
# check for missing values in all the columns
+boston.isnull().sum()
+
+
+
+
+

We can then visualize the data

+
+
+
# set the size of the figure
+sns.set(rc={'figure.figsize':(11.7,8.27)})
+
+# plot a histogram showing the distribution of the target values
+sns.distplot(boston['MEDV'], bins=30)
+plt.show()
+
+
+
+
+

It is now useful to look at the correlation matrix

+
+
+
# compute the pair wise correlation for all columns  
+correlation_matrix = boston.corr().round(2)
+# use the heatmap function from seaborn to plot the correlation matrix
+# annot = True to print the values inside the square
+sns.heatmap(data=correlation_matrix, annot=True)
+
+
+
+
+

From the above coorelation plot we can see that MEDV is strongly correlated to LSTAT and RM. We see also that RAD and TAX are stronly correlated, but we don’t include this in our features together to avoid multi-colinearity

+
+
+
plt.figure(figsize=(20, 5))
+
+features = ['LSTAT', 'RM']
+target = boston['MEDV']
+
+for i, col in enumerate(features):
+    plt.subplot(1, len(features) , i+1)
+    x = boston[col]
+    y = target
+    plt.scatter(x, y, marker='o')
+    plt.title(col)
+    plt.xlabel(col)
+    plt.ylabel('MEDV')
+
+
+
+
+

Now we start training our model

+
+
+
X = pd.DataFrame(np.c_[boston['LSTAT'], boston['RM']], columns = ['LSTAT','RM'])
+Y = boston['MEDV']
+
+
+
+
+

We split the data into training and test sets

+
+
+
from sklearn.model_selection import train_test_split
+
+# splits the training and test data set in 80% : 20%
+# assign random_state to any value.This ensures consistency.
+X_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size = 0.2, random_state=5)
+print(X_train.shape)
+print(X_test.shape)
+print(Y_train.shape)
+print(Y_test.shape)
+
+
+
+
+

Then we use the linear regression functionality from Scikit-Learn

+
+
+
from sklearn.linear_model import LinearRegression
+from sklearn.metrics import mean_squared_error, r2_score
+
+lin_model = LinearRegression()
+lin_model.fit(X_train, Y_train)
+
+# model evaluation for training set
+
+y_train_predict = lin_model.predict(X_train)
+rmse = (np.sqrt(mean_squared_error(Y_train, y_train_predict)))
+r2 = r2_score(Y_train, y_train_predict)
+
+print("The model performance for training set")
+print("--------------------------------------")
+print('RMSE is {}'.format(rmse))
+print('R2 score is {}'.format(r2))
+print("\n")
+
+# model evaluation for testing set
+
+y_test_predict = lin_model.predict(X_test)
+# root mean square error of the model
+rmse = (np.sqrt(mean_squared_error(Y_test, y_test_predict)))
+
+# r-squared score of the model
+r2 = r2_score(Y_test, y_test_predict)
+
+print("The model performance for testing set")
+print("--------------------------------------")
+print('RMSE is {}'.format(rmse))
+print('R2 score is {}'.format(r2))
+
+
+
+
+
+
+
# plotting the y_test vs y_pred
+# ideally should have been a straight line
+plt.scatter(Y_test, y_test_predict)
+plt.show()
+
+
+
+
+
+
+

1.18. Reducing the number of degrees of freedom, overarching view

+

Many Machine Learning problems involve thousands or even millions of +features for each training instance. Not only does this make training +extremely slow, it can also make it much harder to find a good +solution, as we will see. This problem is often referred to as the +curse of dimensionality. Fortunately, in real-world problems, it is +often possible to reduce the number of features considerably, turning +an intractable problem into a tractable one.

+

Later we will discuss some of the most popular dimensionality reduction +techniques: the principal component analysis (PCA), Kernel PCA, and +Locally Linear Embedding (LLE).

+

Principal component analysis and its various variants deal with the +problem of fitting a low-dimensional affine +subspace to a set of of +data points in a high-dimensional space. With its family of methods it +is one of the most used tools in data modeling, compression and +visualization.

+

Before we proceed however, we will discuss how to preprocess our +data. Till now and in connection with our previous examples we have +not met so many cases where we are too sensitive to the scaling of our +data. Normally the data may need a rescaling and/or may be sensitive +to extreme values. Scaling the data renders our inputs much more +suitable for the algorithms we want to employ.

+

Scikit-Learn has several functions which allow us to rescale the +data, normally resulting in much better results in terms of various +accuracy scores. The StandardScaler function in Scikit-Learn +ensures that for each feature/predictor we study the mean value is +zero and the variance is one (every column in the design/feature +matrix). This scaling has the drawback that it does not ensure that +we have a particular maximum or minimum in our data set. Another +function included in Scikit-Learn is the MinMaxScaler which +ensures that all features are exactly between \(0\) and \(1\). The

+

The Normalizer scales each data +point such that the feature vector has a euclidean length of one. In other words, it +projects a data point on the circle (or sphere in the case of higher dimensions) with a +radius of 1. This means every data point is scaled by a different number (by the +inverse of it’s length). +This normalization is often used when only the direction (or angle) of the data matters, +not the length of the feature vector.

+

The RobustScaler works similarly to the StandardScaler in that it +ensures statistical properties for each feature that guarantee that +they are on the same scale. However, the RobustScaler uses the median +and quartiles, instead of mean and variance. This makes the +RobustScaler ignore data points that are very different from the rest +(like measurement errors). These odd data points are also called +outliers, and might often lead to trouble for other scaling +techniques.

+
+

1.18.1. Simple preprocessing examples, Franke function and regression

+
+
+
# Common imports
+import os
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+import sklearn.linear_model as skl
+from sklearn.metrics import mean_squared_error
+from sklearn.model_selection import  train_test_split
+from sklearn.preprocessing import MinMaxScaler, StandardScaler, Normalizer
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+    os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+    os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+    os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+    return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+    return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+    plt.savefig(image_path(fig_id) + ".png", format='png')
+
+
+def FrankeFunction(x,y):
+	term1 = 0.75*np.exp(-(0.25*(9*x-2)**2) - 0.25*((9*y-2)**2))
+	term2 = 0.75*np.exp(-((9*x+1)**2)/49.0 - 0.1*(9*y+1))
+	term3 = 0.5*np.exp(-(9*x-7)**2/4.0 - 0.25*((9*y-3)**2))
+	term4 = -0.2*np.exp(-(9*x-4)**2 - (9*y-7)**2)
+	return term1 + term2 + term3 + term4
+
+
+def create_X(x, y, n ):
+	if len(x.shape) > 1:
+		x = np.ravel(x)
+		y = np.ravel(y)
+
+	N = len(x)
+	l = int((n+1)*(n+2)/2)		# Number of elements in beta
+	X = np.ones((N,l))
+
+	for i in range(1,n+1):
+		q = int((i)*(i+1)/2)
+		for k in range(i+1):
+			X[:,q+k] = (x**(i-k))*(y**k)
+
+	return X
+
+
+# Making meshgrid of datapoints and compute Franke's function
+n = 5
+N = 1000
+x = np.sort(np.random.uniform(0, 1, N))
+y = np.sort(np.random.uniform(0, 1, N))
+z = FrankeFunction(x, y)
+X = create_X(x, y, n=n)    
+# split in training and test data
+X_train, X_test, y_train, y_test = train_test_split(X,z,test_size=0.2)
+
+
+clf = skl.LinearRegression().fit(X_train, y_train)
+
+# The mean squared error and R2 score
+print("MSE before scaling: {:.2f}".format(mean_squared_error(clf.predict(X_test), y_test)))
+print("R2 score before scaling {:.2f}".format(clf.score(X_test,y_test)))
+
+scaler = StandardScaler()
+scaler.fit(X_train)
+X_train_scaled = scaler.transform(X_train)
+X_test_scaled = scaler.transform(X_test)
+
+print("Feature min values before scaling:\n {}".format(X_train.min(axis=0)))
+print("Feature max values before scaling:\n {}".format(X_train.max(axis=0)))
+
+print("Feature min values after scaling:\n {}".format(X_train_scaled.min(axis=0)))
+print("Feature max values after scaling:\n {}".format(X_train_scaled.max(axis=0)))
+
+clf = skl.LinearRegression().fit(X_train_scaled, y_train)
+
+
+print("MSE after  scaling: {:.2f}".format(mean_squared_error(clf.predict(X_test_scaled), y_test)))
+print("R2 score for  scaled data: {:.2f}".format(clf.score(X_test_scaled,y_test)))
+
+
+
+
+
+
+
+ + + + +
+ + + + +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/chapter2.html b/doc/LectureNotes/_build/html/chapter2.html new file mode 100644 index 000000000..71471640c --- /dev/null +++ b/doc/LectureNotes/_build/html/chapter2.html @@ -0,0 +1,1371 @@ + + + + + + + + 2. Resampling Methods — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ +
+
+ +
+ + + + + + + + + + + + + + +
+ + + +
+
+
+
+ +
+ +
+

2. Resampling Methods

+

Video of Lecture

+
+

2.1. Introduction

+

Resampling methods are an indispensable tool in modern +statistics. They involve repeatedly drawing samples from a training +set and refitting a model of interest on each sample in order to +obtain additional information about the fitted model. For example, in +order to estimate the variability of a linear regression fit, we can +repeatedly draw different samples from the training data, fit a linear +regression to each new sample, and then examine the extent to which +the resulting fits differ. Such an approach may allow us to obtain +information that would not be available from fitting the model only +once using the original training sample.

+

Two resampling methods are often used in Machine Learning analyses,

+
    +
  1. The bootstrap method

  2. +
  3. and Cross-Validation

  4. +
+

In addition there are several other methods such as the Jackknife and the Blocking methods. We will discuss in particular +cross-validation and the bootstrap method.

+

Resampling approaches can be computationally expensive, because they +involve fitting the same statistical method multiple times using +different subsets of the training data. However, due to recent +advances in computing power, the computational requirements of +resampling methods generally are not prohibitive. In this chapter, we +discuss two of the most commonly used resampling methods, +cross-validation and the bootstrap. Both methods are important tools +in the practical application of many statistical learning +procedures. For example, cross-validation can be used to estimate the +test error associated with a given statistical learning method in +order to evaluate its performance, or to select the appropriate level +of flexibility. The process of evaluating a model’s performance is +known as model assessment, whereas the process of selecting the proper +level of flexibility for a model is known as model selection. The +bootstrap is widely used.

+
    +
  • Our simulations can be treated as computer experiments. This is particularly the case for Monte Carlo methods

  • +
  • The results can be analysed with the same statistical tools as we would use analysing experimental data.

  • +
  • As in all experiments, we are looking for expectation values and an estimate of how accurate they are, i.e., possible sources for errors.

  • +
+
+
+

2.2. Reminder on Statistics

+
    +
  • As in other experiments, many numerical experiments have two classes of errors:

    +
      +
    • Statistical errors

    • +
    • Systematical errors

    • +
    +
  • +
  • Statistical errors can be estimated using standard tools from statistics

  • +
  • Systematical errors are method specific and must be treated differently from case to case.

  • +
+

The +advantage of doing linear regression is that we actually end up with +analytical expressions for several statistical quantities.
+Standard least squares and Ridge regression allow us to +derive quantities like the variance and other expectation values in a +rather straightforward way.

+

It is assumed that \(\varepsilon_i +\sim \mathcal{N}(0, \sigma^2)\) and the \(\varepsilon_{i}\) are +independent, i.e.:

+
+\[\begin{split} +\begin{align*} +\mbox{Cov}(\varepsilon_{i_1}, +\varepsilon_{i_2}) & = \left\{ \begin{array}{lcc} \sigma^2 & \mbox{if} +& i_1 = i_2, \\ 0 & \mbox{if} & i_1 \not= i_2. \end{array} \right. +\end{align*} +\end{split}\]
+

The randomness of \(\varepsilon_i\) implies that +\(\mathbf{y}_i\) is also a random variable. In particular, +\(\mathbf{y}_i\) is normally distributed, because \(\varepsilon_i \sim +\mathcal{N}(0, \sigma^2)\) and \(\mathbf{X}_{i,\ast} \, \boldsymbol{\beta}\) is a +non-random scalar. To specify the parameters of the distribution of +\(\mathbf{y}_i\) we need to calculate its first two moments.

+

Recall that \(\boldsymbol{X}\) is a matrix of dimensionality \(n\times p\). The +notation above \(\mathbf{X}_{i,\ast}\) means that we are looking at the +row number \(i\) and perform a sum over all values \(p\).

+

The assumption we have made here can be summarized as (and this is going to be useful when we discuss the bias-variance trade off) +that there exists a function \(f(\boldsymbol{x})\) and a normal distributed error \(\boldsymbol{\varepsilon}\sim \mathcal{N}(0, \sigma^2)\) +which describe our data

+
+\[ +\boldsymbol{y} = f(\boldsymbol{x})+\boldsymbol{\varepsilon} +\]
+

We approximate this function with our model from the solution of the linear regression equations, that is our +function \(f\) is approximated by \(\boldsymbol{\tilde{y}}\) where we want to minimize \((\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\), our MSE, with

+
+\[ +\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\beta}. +\]
+

We can calculate the expectation value of \(\boldsymbol{y}\) for a given element \(i\)

+
+\[ +\begin{align*} +\mathbb{E}(y_i) & = +\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\beta}) + \mathbb{E}(\varepsilon_i) +\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \beta, +\end{align*} +\]
+

while +its variance is

+
+\[\begin{split} +\begin{align*} \mbox{Var}(y_i) & = \mathbb{E} \{ [y_i +- \mathbb{E}(y_i)]^2 \} \, \, \, = \, \, \, \mathbb{E} ( y_i^2 ) - +[\mathbb{E}(y_i)]^2 \\ & = \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, +\beta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 \\ & += \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 \varepsilon_i +\mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, +\ast} \, \beta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 +\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + +\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 +\\ & = \mathbb{E}(\varepsilon_i^2 ) \, \, \, = \, \, \, +\mbox{Var}(\varepsilon_i) \, \, \, = \, \, \, \sigma^2. +\end{align*} +\end{split}\]
+

Hence, \(y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta}, \sigma^2)\), that is \(\boldsymbol{y}\) follows a normal distribution with +mean value \(\boldsymbol{X}\boldsymbol{\beta}\) and variance \(\sigma^2\) (not be confused with the singular values of the SVD).

+

With the OLS expressions for the parameters \(\boldsymbol{\beta}\) we can evaluate the expectation value

+
+\[ +\mathbb{E}(\boldsymbol{\beta}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\beta}=\boldsymbol{\beta}. +\]
+

This means that the estimator of the regression parameters is unbiased.

+

We can also calculate the variance

+

The variance of \(\boldsymbol{\beta}\) is

+
+\[\begin{split} +\begin{eqnarray*} +\mbox{Var}(\boldsymbol{\beta}) & = & \mathbb{E} \{ [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})] [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})]^{T} \} +\\ +& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}]^{T} \} +\\ +% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% \\ +% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% \\ +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +\\ +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% \\ +% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} +% \\ +% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\beta} \boldsymbol{\beta}^T +\\ +& = & \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +\, \, \, = \, \, \, \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}, +\end{eqnarray*} +\end{split}\]
+

where we have used that \(\mathbb{E} (\mathbf{Y} \mathbf{Y}^{T}) = +\mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + +\sigma^2 \, \mathbf{I}_{nn}\). From \(\mbox{Var}(\boldsymbol{\beta}) = \sigma^2 +\, (\mathbf{X}^{T} \mathbf{X})^{-1}\), one obtains an estimate of the +variance of the estimate of the \(j\)-th regression coefficient: +\(\boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 \sqrt{ +[(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} }\). This may be used to +construct a confidence interval for the estimates.

+

In a similar way, we can obtain analytical expressions for say the +expectation values of the parameters \(\boldsymbol{\beta}\) and their variance +when we employ Ridge regression, allowing us again to define a confidence interval.

+

It is rather straightforward to show that

+
+\[ +\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\beta}^{\mathrm{OLS}}. +\]
+

We see clearly that +\(\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\beta}^{\mathrm{OLS}}\) for any \(\lambda > 0\). We say then that the ridge estimator is biased.

+

We can also compute the variance as

+
+\[ +\mbox{Var}[\boldsymbol{\beta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, +\]
+

and it is easy to see that if the parameter \(\lambda\) goes to infinity then the variance of Ridge parameters \(\boldsymbol{\beta}\) goes to zero.

+

With this, we can compute the difference

+
+\[ +\mbox{Var}[\boldsymbol{\beta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\beta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. +\]
+

The difference is non-negative definite since each component of the +matrix product is non-negative definite. +This means the variance we obtain with the standard OLS will always for \(\lambda > 0\) be larger than the variance of \(\boldsymbol{\beta}\) obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below.

+
+
+

2.3. Resampling methods

+

With all these analytical equations for both the OLS and Ridge +regression, we will now outline how to assess a given model. This will +lead us to a discussion of the so-called bias-variance tradeoff (see +below) and so-called resampling methods.

+

One of the quantities we have discussed as a way to measure errors is +the mean-squared error (MSE), mainly used for fitting of continuous +functions. Another choice is the absolute error.

+

In the discussions below we will focus on the MSE and in particular since we will split the data into test and training data, +we discuss the

+
    +
  1. prediction error or simply the test error \(\mathrm{Err_{Test}}\), where we have a fixed training set and the test error is the MSE arising from the data reserved for testing. We discuss also the

  2. +
  3. training error \(\mathrm{Err_{Train}}\), which is the average loss over the training data.

  4. +
+

As our model becomes more and more complex, more of the training data tends to used. The training may thence adapt to more complicated structures in the data. This may lead to a decrease in the bias (see below for code example) and a slight increase of the variance for the test error. +For a certain level of complexity the test error will reach minimum, before starting to increase again. The +training error reaches a saturation.

+

Two famous +resampling methods are the independent bootstrap and the jackknife.

+

The jackknife is a special case of the independent bootstrap. Still, the jackknife was made +popular prior to the independent bootstrap. And as the popularity of +the independent bootstrap soared, new variants, such as the dependent bootstrap.

+

The Jackknife and independent bootstrap work for +independent, identically distributed random variables. +If these conditions are not +satisfied, the methods will fail. Yet, it should be said that if the data are +independent, identically distributed, and we only want to estimate the +variance of \(\overline{X}\) (which often is the case), then there is no +need for bootstrapping.

+

The Jackknife works by making many replicas of the estimator \(\widehat{\theta}\). +The jackknife is a resampling method where we systematically leave out one observation from the vector of observed values \(\boldsymbol{x} = (x_1,x_2,\cdots,X_n)\). +Let \(\boldsymbol{x}_i\) denote the vector

+
+\[ +\boldsymbol{x}_i = (x_1,x_2,\cdots,x_{i-1},x_{i+1},\cdots,x_n), +\]
+

which equals the vector \(\boldsymbol{x}\) with the exception that observation +number \(i\) is left out. Using this notation, define +\(\widehat{\theta}_i\) to be the estimator +\(\widehat{\theta}\) computed using \(\vec{X}_i\).

+
+
+
from numpy import *
+from numpy.random import randint, randn
+from time import time
+
+def jackknife(data, stat):
+    n = len(data);t = zeros(n); inds = arange(n); t0 = time()
+    ## 'jackknifing' by leaving out an observation for each i                                                                                                                      
+    for i in range(n):
+        t[i] = stat(delete(data,i) )
+
+    # analysis                                                                                                                                                                     
+    print("Runtime: %g sec" % (time()-t0)); print("Jackknife Statistics :")
+    print("original           bias      std. error")
+    print("%8g %14g %15g" % (stat(data),(n-1)*mean(t)/n, (n*var(t))**.5))
+
+    return t
+
+
+# Returns mean of data samples                                                                                                                                                     
+def stat(data):
+    return mean(data)
+
+
+mu, sigma = 100, 15
+datapoints = 10000
+x = mu + sigma*random.randn(datapoints)
+# jackknife returns the data sample                                                                                                                                                
+t = jackknife(x, stat)
+
+
+
+
+
Runtime: 0.136239 sec
+Jackknife Statistics :
+original           bias      std. error
+ 99.9154        99.9054        0.150467
+
+
+
+
+
+

2.3.1. Bootstrap

+

Bootstrapping is a nonparametric approach to statistical inference +that substitutes computation for more traditional distributional +assumptions and asymptotic results. Bootstrapping offers a number of +advantages:

+
    +
  1. The bootstrap is quite general, although there are some cases in which it fails.

  2. +
  3. Because it does not require distributional assumptions (such as normally distributed errors), the bootstrap can provide more accurate inferences when the data are not well behaved or when the sample size is small.

  4. +
  5. It is possible to apply the bootstrap to statistics with sampling distributions that are difficult to derive, even asymptotically.

  6. +
  7. It is relatively simple to apply the bootstrap to complex data-collection plans (such as stratified and clustered samples).

  8. +
+

Since \(\widehat{\theta} = \widehat{\theta}(\boldsymbol{X})\) is a function of random variables, +\(\widehat{\theta}\) itself must be a random variable. Thus it has +a pdf, call this function \(p(\boldsymbol{t})\). The aim of the bootstrap is to +estimate \(p(\boldsymbol{t})\) by the relative frequency of +\(\widehat{\theta}\). You can think of this as using a histogram +in the place of \(p(\boldsymbol{t})\). If the relative frequency closely +resembles \(p(\vec{t})\), then using numerics, it is straight forward to +estimate all the interesting parameters of \(p(\boldsymbol{t})\) using point +estimators.

+

In the case that \(\widehat{\theta}\) has +more than one component, and the components are independent, we use the +same estimator on each component separately. If the probability +density function of \(X_i\), \(p(x)\), had been known, then it would have +been straight forward to do this by:

+
    +
  1. Drawing lots of numbers from \(p(x)\), suppose we call one such set of numbers \((X_1^*, X_2^*, \cdots, X_n^*)\).

  2. +
  3. Then using these numbers, we could compute a replica of \(\widehat{\theta}\) called \(\widehat{\theta}^*\).

  4. +
+

By repeated use of (1) and (2), many +estimates of \(\widehat{\theta}\) could have been obtained. The +idea is to use the relative frequency of \(\widehat{\theta}^*\) +(think of a histogram) as an estimate of \(p(\boldsymbol{t})\).

+

But +unless there is enough information available about the process that +generated \(X_1,X_2,\cdots,X_n\), \(p(x)\) is in general +unknown. Therefore, Efron in 1979 asked the +question: What if we replace \(p(x)\) by the relative frequency +of the observation \(X_i\); if we draw observations in accordance with +the relative frequency of the observations, will we obtain the same +result in some asymptotic sense? The answer is yes.

+

Instead of generating the histogram for the relative +frequency of the observation \(X_i\), just draw the values +\((X_1^*,X_2^*,\cdots,X_n^*)\) with replacement from the vector +\(\boldsymbol{X}\).

+

The independent bootstrap works like this:

+
    +
  1. Draw with replacement \(n\) numbers for the observed variables \(\boldsymbol{x} = (x_1,x_2,\cdots,x_n)\).

  2. +
  3. Define a vector \(\boldsymbol{x}^*\) containing the values which were drawn from \(\boldsymbol{x}\).

  4. +
  5. Using the vector \(\boldsymbol{x}^*\) compute \(\widehat{\theta}^*\) by evaluating \(\widehat \theta\) under the observations \(\boldsymbol{x}^*\).

  6. +
  7. Repeat this process \(k\) times.

  8. +
+

When you are done, you can draw a histogram of the relative frequency +of \(\widehat \theta^*\). This is your estimate of the probability +distribution \(p(t)\). Using this probability distribution you can +estimate any statistics thereof. In principle you never draw the +histogram of the relative frequency of \(\widehat{\theta}^*\). Instead +you use the estimators corresponding to the statistic of interest. For +example, if you are interested in estimating the variance of \(\widehat +\theta\), apply the etsimator \(\widehat \sigma^2\) to the values +\(\widehat \theta ^*\).

+

The following code starts with a Gaussian distribution with mean value +\(\mu =100\) and variance \(\sigma=15\). We use this to generate the data +used in the bootstrap analysis. The bootstrap analysis returns a data +set after a given number of bootstrap operations (as many as we have +data points). This data set consists of estimated mean values for each +bootstrap operation. The histogram generated by the bootstrap method +shows that the distribution for these mean values is also a Gaussian, +centered around the mean value \(\mu=100\) but with standard deviation +\(\sigma/\sqrt{n}\), where \(n\) is the number of bootstrap samples (in +this case the same as the number of original data points). The value +of the standard deviation is what we expect from the central limit +theorem.

+
+
+
%matplotlib inline
+
+from numpy import *
+from numpy.random import randint, randn
+from time import time
+import matplotlib.mlab as mlab
+import matplotlib.pyplot as plt
+
+# Returns mean of bootstrap samples                                                                                                                                                
+def stat(data):
+    return mean(data)
+
+# Bootstrap algorithm
+def bootstrap(data, statistic, R):
+    t = zeros(R); n = len(data); inds = arange(n); t0 = time()
+    # non-parametric bootstrap         
+    for i in range(R):
+        t[i] = statistic(data[randint(0,n,n)])
+
+    # analysis    
+    print("Runtime: %g sec" % (time()-t0)); print("Bootstrap Statistics :")
+    print("original           bias      std. error")
+    print("%8g %8g %14g %15g" % (statistic(data), std(data),mean(t),std(t)))
+    return t
+
+
+mu, sigma = 100, 15
+datapoints = 10000
+x = mu + sigma*random.randn(datapoints)
+# bootstrap returns the data sample                                    
+t = bootstrap(x, stat, datapoints)
+# the histogram of the bootstrapped  data                                                                                                    
+n, binsboot, patches = plt.hist(t, 50, normed=1, facecolor='red', alpha=0.75)
+
+# add a 'best fit' line  
+y = mlab.normpdf( binsboot, mean(t), std(t))
+lt = plt.plot(binsboot, y, 'r--', linewidth=1)
+plt.xlabel('Smarts')
+plt.ylabel('Probability')
+plt.axis([99.5, 100.6, 0, 3.0])
+plt.grid(True)
+
+plt.show()
+
+
+
+
+
Runtime: 1.77593 sec
+Bootstrap Statistics :
+original           bias      std. error
+ 100.207  14.9646        100.209        0.150205
+
+
+
---------------------------------------------------------------------------
+AttributeError                            Traceback (most recent call last)
+<ipython-input-2-772b904ae9cb> in <module>
+     31 t = bootstrap(x, stat, datapoints)
+     32 # the histogram of the bootstrapped  data
+---> 33 n, binsboot, patches = plt.hist(t, 50, normed=1, facecolor='red', alpha=0.75)
+     34 
+     35 # add a 'best fit' line
+
+~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/pyplot.py in hist(x, bins, range, density, weights, cumulative, bottom, histtype, align, orientation, rwidth, log, color, label, stacked, data, **kwargs)
+   2683         orientation='vertical', rwidth=None, log=False, color=None,
+   2684         label=None, stacked=False, *, data=None, **kwargs):
+-> 2685     return gca().hist(
+   2686         x, bins=bins, range=range, density=density, weights=weights,
+   2687         cumulative=cumulative, bottom=bottom, histtype=histtype,
+
+~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/__init__.py in inner(ax, data, *args, **kwargs)
+   1445     def inner(ax, *args, data=None, **kwargs):
+   1446         if data is None:
+-> 1447             return func(ax, *map(sanitize_sequence, args), **kwargs)
+   1448 
+   1449         bound = new_sig.bind(ax, *args, **kwargs)
+
+~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/axes/_axes.py in hist(self, x, bins, range, density, weights, cumulative, bottom, histtype, align, orientation, rwidth, log, color, label, stacked, **kwargs)
+   6813             if patch:
+   6814                 p = patch[0]
+-> 6815                 p.update(kwargs)
+   6816                 if lbl is not None:
+   6817                     p.set_label(lbl)
+
+~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/artist.py in update(self, props)
+    994                     func = getattr(self, f"set_{k}", None)
+    995                     if not callable(func):
+--> 996                         raise AttributeError(f"{type(self).__name__!r} object "
+    997                                              f"has no property {k!r}")
+    998                     ret.append(func(v))
+
+AttributeError: 'Rectangle' object has no property 'normed'
+
+
+_images/chapter2_25_2.png +
+
+
+
+
+

2.4. Various steps in cross-validation

+

When the repetitive splitting of the data set is done randomly, +samples may accidently end up in a fast majority of the splits in +either training or test set. Such samples may have an unbalanced +influence on either model building or prediction evaluation. To avoid +this \(k\)-fold cross-validation structures the data splitting. The +samples are divided into \(k\) more or less equally sized exhaustive and +mutually exclusive subsets. In turn (at each split) one of these +subsets plays the role of the test set while the union of the +remaining subsets constitutes the training set. Such a splitting +warrants a balanced representation of each sample in both training and +test set over the splits. Still the division into the \(k\) subsets +involves a degree of randomness. This may be fully excluded when +choosing \(k=n\). This particular case is referred to as leave-one-out +cross-validation (LOOCV).

+
    +
  • Define a range of interest for the penalty parameter.

  • +
  • Divide the data set into training and test set comprising samples \(\{1, \ldots, n\} \setminus i\) and \(\{ i \}\), respectively.

  • +
  • Fit the linear regression model by means of ridge estimation for each \(\lambda\) in the grid using the training set, and the corresponding estimate of the error variance \(\boldsymbol{\sigma}_{-i}^2(\lambda)\), as

  • +
+
+\[ +\begin{align*} +\boldsymbol{\beta}_{-i}(\lambda) & = ( \boldsymbol{X}_{-i, \ast}^{T} +\boldsymbol{X}_{-i, \ast} + \lambda \boldsymbol{I}_{pp})^{-1} +\boldsymbol{X}_{-i, \ast}^{T} \boldsymbol{y}_{-i} +\end{align*} +\]
+
    +
  • Evaluate the prediction performance of these models on the test set by \(\log\{L[y_i, \boldsymbol{X}_{i, \ast}; \boldsymbol{\beta}_{-i}(\lambda), \boldsymbol{\sigma}_{-i}^2(\lambda)]\}\). Or, by the prediction error \(|y_i - \boldsymbol{X}_{i, \ast} \boldsymbol{\beta}_{-i}(\lambda)|\), the relative error, the error squared or the R2 score function.

  • +
  • Repeat the first three steps such that each sample plays the role of the test set once.

  • +
  • Average the prediction performances of the test sets at each grid point of the penalty bias/parameter. It is an estimate of the prediction performance of the model corresponding to this value of the penalty parameter on novel data. It is defined as

  • +
+
+\[ +\begin{align*} +\frac{1}{n} \sum_{i = 1}^n \log\{L[y_i, \mathbf{X}_{i, \ast}; \boldsymbol{\beta}_{-i}(\lambda), \boldsymbol{\sigma}_{-i}^2(\lambda)]\}. +\end{align*} +\]
+

For the various values of \(k\)

+
    +
  1. shuffle the dataset randomly.

  2. +
  3. Split the dataset into \(k\) groups.

  4. +
  5. For each unique group:

  6. +
+

a. Decide which group to use as set for test data

+

b. Take the remaining groups as a training data set

+

c. Fit a model on the training set and evaluate it on the test set

+

d. Retain the evaluation score and discard the model

+
    +
  1. Summarize the model using the sample of model evaluation scores

  2. +
+

The code here uses Ridge regression with cross-validation (CV) resampling and \(k\)-fold CV in order to fit a specific polynomial.

+
+
+
import numpy as np
+import matplotlib.pyplot as plt
+from sklearn.model_selection import KFold
+from sklearn.linear_model import Ridge
+from sklearn.model_selection import cross_val_score
+from sklearn.preprocessing import PolynomialFeatures
+
+# A seed just to ensure that the random numbers are the same for every run.
+# Useful for eventual debugging.
+np.random.seed(3155)
+
+# Generate the data.
+nsamples = 100
+x = np.random.randn(nsamples)
+y = 3*x**2 + np.random.randn(nsamples)
+
+## Cross-validation on Ridge regression using KFold only
+
+# Decide degree on polynomial to fit
+poly = PolynomialFeatures(degree = 6)
+
+# Decide which values of lambda to use
+nlambdas = 500
+lambdas = np.logspace(-3, 5, nlambdas)
+
+# Initialize a KFold instance
+k = 5
+kfold = KFold(n_splits = k)
+
+# Perform the cross-validation to estimate MSE
+scores_KFold = np.zeros((nlambdas, k))
+
+i = 0
+for lmb in lambdas:
+    ridge = Ridge(alpha = lmb)
+    j = 0
+    for train_inds, test_inds in kfold.split(x):
+        xtrain = x[train_inds]
+        ytrain = y[train_inds]
+
+        xtest = x[test_inds]
+        ytest = y[test_inds]
+
+        Xtrain = poly.fit_transform(xtrain[:, np.newaxis])
+        ridge.fit(Xtrain, ytrain[:, np.newaxis])
+
+        Xtest = poly.fit_transform(xtest[:, np.newaxis])
+        ypred = ridge.predict(Xtest)
+
+        scores_KFold[i,j] = np.sum((ypred - ytest[:, np.newaxis])**2)/np.size(ypred)
+
+        j += 1
+    i += 1
+
+
+estimated_mse_KFold = np.mean(scores_KFold, axis = 1)
+
+## Cross-validation using cross_val_score from sklearn along with KFold
+
+# kfold is an instance initialized above as:
+# kfold = KFold(n_splits = k)
+
+estimated_mse_sklearn = np.zeros(nlambdas)
+i = 0
+for lmb in lambdas:
+    ridge = Ridge(alpha = lmb)
+
+    X = poly.fit_transform(x[:, np.newaxis])
+    estimated_mse_folds = cross_val_score(ridge, X, y[:, np.newaxis], scoring='neg_mean_squared_error', cv=kfold)
+
+    # cross_val_score return an array containing the estimated negative mse for every fold.
+    # we have to the the mean of every array in order to get an estimate of the mse of the model
+    estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
+
+    i += 1
+
+## Plot and compare the slightly different ways to perform cross-validation
+
+plt.figure()
+
+plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
+plt.plot(np.log10(lambdas), estimated_mse_KFold, 'r--', label = 'KFold')
+
+plt.xlabel('log10(lambda)')
+plt.ylabel('mse')
+
+plt.legend()
+
+plt.show()
+
+
+
+
+
+
+

2.5. The bias-variance tradeoff

+

We will discuss the bias-variance tradeoff in the context of +continuous predictions such as regression. However, many of the +intuitions and ideas discussed here also carry over to classification +tasks. Consider a dataset \(\mathcal{L}\) consisting of the data +\(\mathbf{X}_\mathcal{L}=\{(y_j, \boldsymbol{x}_j), j=0\ldots n-1\}\).

+

Let us assume that the true data is generated from a noisy model

+
+\[ +\boldsymbol{y}=f(\boldsymbol{x}) + \boldsymbol{\epsilon} +\]
+

where \(\epsilon\) is normally distributed with mean zero and standard deviation \(\sigma^2\).

+

In our derivation of the ordinary least squares method we defined then +an approximation to the function \(f\) in terms of the parameters +\(\boldsymbol{\beta}\) and the design matrix \(\boldsymbol{X}\) which embody our model, +that is \(\boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\beta}\).

+

Thereafter we found the parameters \(\boldsymbol{\beta}\) by optimizing the means squared error via the so-called cost function

+
+\[ +C(\boldsymbol{X},\boldsymbol{\beta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. +\]
+

We can rewrite this as

+
+\[ +\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]=\frac{1}{n}\sum_i(f_i-\mathbb{E}\left[\boldsymbol{\tilde{y}}\right])^2+\frac{1}{n}\sum_i(\tilde{y}_i-\mathbb{E}\left[\boldsymbol{\tilde{y}}\right])^2+\sigma^2. +\]
+

The three terms represent the square of the bias of the learning +method, which can be thought of as the error caused by the simplifying +assumptions built into the method. The second term represents the +variance of the chosen model and finally the last terms is variance of +the error \(\boldsymbol{\epsilon}\).

+

To derive this equation, we need to recall that the variance of \(\boldsymbol{y}\) and \(\boldsymbol{\epsilon}\) are both equal to \(\sigma^2\). The mean value of \(\boldsymbol{\epsilon}\) is by definition equal to zero. Furthermore, the function \(f\) is not a stochastics variable, idem for \(\boldsymbol{\tilde{y}}\). +We use a more compact notation in terms of the expectation value

+
+\[ +\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]=\mathbb{E}\left[(\boldsymbol{f}+\boldsymbol{\epsilon}-\boldsymbol{\tilde{y}})^2\right], +\]
+

and adding and subtracting \(\mathbb{E}\left[\boldsymbol{\tilde{y}}\right]\) we get

+
+\[ +\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]=\mathbb{E}\left[(\boldsymbol{f}+\boldsymbol{\epsilon}-\boldsymbol{\tilde{y}}+\mathbb{E}\left[\boldsymbol{\tilde{y}}\right]-\mathbb{E}\left[\boldsymbol{\tilde{y}}\right])^2\right], +\]
+

which, using the abovementioned expectation values can be rewritten as

+
+\[ +\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]=\mathbb{E}\left[(\boldsymbol{y}-\mathbb{E}\left[\boldsymbol{\tilde{y}}\right])^2\right]+\mathrm{Var}\left[\boldsymbol{\tilde{y}}\right]+\sigma^2, +\]
+

that is the rewriting in terms of the so-called bias, the variance of the model \(\boldsymbol{\tilde{y}}\) and the variance of \(\boldsymbol{\epsilon}\).

+
+
+
import matplotlib.pyplot as plt
+import numpy as np
+from sklearn.linear_model import LinearRegression, Ridge, Lasso
+from sklearn.preprocessing import PolynomialFeatures
+from sklearn.model_selection import train_test_split
+from sklearn.pipeline import make_pipeline
+from sklearn.utils import resample
+
+np.random.seed(2018)
+
+n = 500
+n_boostraps = 100
+degree = 18  # A quite high value, just to show.
+noise = 0.1
+
+# Make data set.
+x = np.linspace(-1, 3, n).reshape(-1, 1)
+y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) + np.random.normal(0, 0.1, x.shape)
+
+# Hold out some test data that is never used in training.
+x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2)
+
+# Combine x transformation and model into one operation.
+# Not neccesary, but convenient.
+model = make_pipeline(PolynomialFeatures(degree=degree), LinearRegression(fit_intercept=False))
+
+# The following (m x n_bootstraps) matrix holds the column vectors y_pred
+# for each bootstrap iteration.
+y_pred = np.empty((y_test.shape[0], n_boostraps))
+for i in range(n_boostraps):
+    x_, y_ = resample(x_train, y_train)
+
+    # Evaluate the new model on the same test data each time.
+    y_pred[:, i] = model.fit(x_, y_).predict(x_test).ravel()
+
+# Note: Expectations and variances taken w.r.t. different training
+# data sets, hence the axis=1. Subsequent means are taken across the test data
+# set in order to obtain a total value, but before this we have error/bias/variance
+# calculated per data point in the test set.
+# Note 2: The use of keepdims=True is important in the calculation of bias as this 
+# maintains the column vector form. Dropping this yields very unexpected results.
+error = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) )
+bias = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 )
+variance = np.mean( np.var(y_pred, axis=1, keepdims=True) )
+print('Error:', error)
+print('Bias^2:', bias)
+print('Var:', variance)
+print('{} >= {} + {} = {}'.format(error, bias, variance, bias+variance))
+
+plt.plot(x[::5, :], y[::5, :], label='f(x)')
+plt.scatter(x_test, y_test, label='Data points')
+plt.scatter(x_test, np.mean(y_pred, axis=1), label='Pred')
+plt.legend()
+plt.show()
+
+
+
+
+
+
+
import matplotlib.pyplot as plt
+import numpy as np
+from sklearn.linear_model import LinearRegression, Ridge, Lasso
+from sklearn.preprocessing import PolynomialFeatures
+from sklearn.model_selection import train_test_split
+from sklearn.pipeline import make_pipeline
+from sklearn.utils import resample
+
+np.random.seed(2018)
+
+n = 40
+n_boostraps = 100
+maxdegree = 14
+
+
+# Make data set.
+x = np.linspace(-3, 3, n).reshape(-1, 1)
+y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
+error = np.zeros(maxdegree)
+bias = np.zeros(maxdegree)
+variance = np.zeros(maxdegree)
+polydegree = np.zeros(maxdegree)
+x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2)
+
+for degree in range(maxdegree):
+    model = make_pipeline(PolynomialFeatures(degree=degree), LinearRegression(fit_intercept=False))
+    y_pred = np.empty((y_test.shape[0], n_boostraps))
+    for i in range(n_boostraps):
+        x_, y_ = resample(x_train, y_train)
+        y_pred[:, i] = model.fit(x_, y_).predict(x_test).ravel()
+
+    polydegree[degree] = degree
+    error[degree] = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) )
+    bias[degree] = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 )
+    variance[degree] = np.mean( np.var(y_pred, axis=1, keepdims=True) )
+    print('Polynomial degree:', degree)
+    print('Error:', error[degree])
+    print('Bias^2:', bias[degree])
+    print('Var:', variance[degree])
+    print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree]))
+
+plt.plot(polydegree, error, label='Error')
+plt.plot(polydegree, bias, label='bias')
+plt.plot(polydegree, variance, label='Variance')
+plt.legend()
+plt.show()
+
+
+
+
+

The bias-variance tradeoff summarizes the fundamental tension in +machine learning, particularly supervised learning, between the +complexity of a model and the amount of training data needed to train +it. Since data is often limited, in practice it is often useful to +use a less-complex model with higher bias, that is a model whose asymptotic +performance is worse than another model because it is easier to +train and less sensitive to sampling noise arising from having a +finite-sized training dataset (smaller variance).

+

The above equations tell us that in +order to minimize the expected test error, we need to select a +statistical learning method that simultaneously achieves low variance +and low bias. Note that variance is inherently a nonnegative quantity, +and squared bias is also nonnegative. Hence, we see that the expected +test MSE can never lie below \(Var(\epsilon)\), the irreducible error.

+

What do we mean by the variance and bias of a statistical learning +method? The variance refers to the amount by which our model would change if we +estimated it using a different training data set. Since the training +data are used to fit the statistical learning method, different +training data sets will result in a different estimate. But ideally the +estimate for our model should not vary too much between training +sets. However, if a method has high variance then small changes in +the training data can result in large changes in the model. In general, more +flexible statistical methods have higher variance.

+

You may also find this recent article of interest.

+
+
+
"""
+============================
+Underfitting vs. Overfitting
+============================
+
+This example demonstrates the problems of underfitting and overfitting and
+how we can use linear regression with polynomial features to approximate
+nonlinear functions. The plot shows the function that we want to approximate,
+which is a part of the cosine function. In addition, the samples from the
+real function and the approximations of different models are displayed. The
+models have polynomial features of different degrees. We can see that a
+linear function (polynomial with degree 1) is not sufficient to fit the
+training samples. This is called **underfitting**. A polynomial of degree 4
+approximates the true function almost perfectly. However, for higher degrees
+the model will **overfit** the training data, i.e. it learns the noise of the
+training data.
+We evaluate quantitatively **overfitting** / **underfitting** by using
+cross-validation. We calculate the mean squared error (MSE) on the validation
+set, the higher, the less likely the model generalizes correctly from the
+training data.
+"""
+
+print(__doc__)
+
+import numpy as np
+import matplotlib.pyplot as plt
+from sklearn.pipeline import Pipeline
+from sklearn.preprocessing import PolynomialFeatures
+from sklearn.linear_model import LinearRegression
+from sklearn.model_selection import cross_val_score
+
+
+def true_fun(X):
+    return np.cos(1.5 * np.pi * X)
+
+np.random.seed(0)
+
+n_samples = 30
+degrees = [1, 4, 15]
+
+X = np.sort(np.random.rand(n_samples))
+y = true_fun(X) + np.random.randn(n_samples) * 0.1
+
+plt.figure(figsize=(14, 5))
+for i in range(len(degrees)):
+    ax = plt.subplot(1, len(degrees), i + 1)
+    plt.setp(ax, xticks=(), yticks=())
+
+    polynomial_features = PolynomialFeatures(degree=degrees[i],
+                                             include_bias=False)
+    linear_regression = LinearRegression()
+    pipeline = Pipeline([("polynomial_features", polynomial_features),
+                         ("linear_regression", linear_regression)])
+    pipeline.fit(X[:, np.newaxis], y)
+
+    # Evaluate the models using crossvalidation
+    scores = cross_val_score(pipeline, X[:, np.newaxis], y,
+                             scoring="neg_mean_squared_error", cv=10)
+
+    X_test = np.linspace(0, 1, 100)
+    plt.plot(X_test, pipeline.predict(X_test[:, np.newaxis]), label="Model")
+    plt.plot(X_test, true_fun(X_test), label="True function")
+    plt.scatter(X, y, edgecolor='b', s=20, label="Samples")
+    plt.xlabel("x")
+    plt.ylabel("y")
+    plt.xlim((0, 1))
+    plt.ylim((-2, 2))
+    plt.legend(loc="best")
+    plt.title("Degree {}\nMSE = {:.2e}(+/- {:.2e})".format(
+        degrees[i], -scores.mean(), scores.std()))
+plt.show()
+
+
+
+
+
+
+
# Common imports
+import os
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.linear_model import LinearRegression, Ridge, Lasso
+from sklearn.model_selection import train_test_split
+from sklearn.utils import resample
+from sklearn.metrics import mean_squared_error
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+    os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+    os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+    os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+    return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+    return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+    plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("EoS.csv"),'r')
+
+# Read the EoS data as  csv file and organize the data into two arrays with density and energies
+EoS = pd.read_csv(infile, names=('Density', 'Energy'))
+EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')
+EoS = EoS.dropna()
+Energies = EoS['Energy']
+Density = EoS['Density']
+#  The design matrix now as function of various polytrops
+
+Maxpolydegree = 30
+X = np.zeros((len(Density),Maxpolydegree))
+X[:,0] = 1.0
+testerror = np.zeros(Maxpolydegree)
+trainingerror = np.zeros(Maxpolydegree)
+polynomial = np.zeros(Maxpolydegree)
+
+trials = 100
+for polydegree in range(1, Maxpolydegree):
+    polynomial[polydegree] = polydegree
+    for degree in range(polydegree):
+        X[:,degree] = Density**(degree/3.0)
+
+# loop over trials in order to estimate the expectation value of the MSE
+    testerror[polydegree] = 0.0
+    trainingerror[polydegree] = 0.0
+    for samples in range(trials):
+        x_train, x_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2)
+        model = LinearRegression(fit_intercept=True).fit(x_train, y_train)
+        ypred = model.predict(x_train)
+        ytilde = model.predict(x_test)
+        testerror[polydegree] += mean_squared_error(y_test, ytilde)
+        trainingerror[polydegree] += mean_squared_error(y_train, ypred) 
+
+    testerror[polydegree] /= trials
+    trainingerror[polydegree] /= trials
+    print("Degree of polynomial: %3d"% polynomial[polydegree])
+    print("Mean squared error on training data: %.8f" % trainingerror[polydegree])
+    print("Mean squared error on test data: %.8f" % testerror[polydegree])
+
+plt.plot(polynomial, np.log10(trainingerror), label='Training Error')
+plt.plot(polynomial, np.log10(testerror), label='Test Error')
+plt.xlabel('Polynomial degree')
+plt.ylabel('log10[MSE]')
+plt.legend()
+plt.show()
+
+
+
+
+
+
+
# Common imports
+import os
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.linear_model import LinearRegression, Ridge, Lasso
+from sklearn.metrics import mean_squared_error
+from sklearn.model_selection import KFold
+from sklearn.model_selection import cross_val_score
+
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+    os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+    os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+    os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+    return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+    return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+    plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("EoS.csv"),'r')
+
+# Read the EoS data as  csv file and organize the data into two arrays with density and energies
+EoS = pd.read_csv(infile, names=('Density', 'Energy'))
+EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')
+EoS = EoS.dropna()
+Energies = EoS['Energy']
+Density = EoS['Density']
+#  The design matrix now as function of various polytrops
+
+Maxpolydegree = 30
+X = np.zeros((len(Density),Maxpolydegree))
+X[:,0] = 1.0
+estimated_mse_sklearn = np.zeros(Maxpolydegree)
+polynomial = np.zeros(Maxpolydegree)
+k =5
+kfold = KFold(n_splits = k)
+
+for polydegree in range(1, Maxpolydegree):
+    polynomial[polydegree] = polydegree
+    for degree in range(polydegree):
+        X[:,degree] = Density**(degree/3.0)
+        OLS = LinearRegression()
+# loop over trials in order to estimate the expectation value of the MSE
+    estimated_mse_folds = cross_val_score(OLS, X, Energies, scoring='neg_mean_squared_error', cv=kfold)
+#[:, np.newaxis]
+    estimated_mse_sklearn[polydegree] = np.mean(-estimated_mse_folds)
+
+plt.plot(polynomial, np.log10(estimated_mse_sklearn), label='Test Error')
+plt.xlabel('Polynomial degree')
+plt.ylabel('log10[MSE]')
+plt.legend()
+plt.show()
+
+
+
+
+
+
+
import numpy as np
+import matplotlib.pyplot as plt
+from sklearn.model_selection import KFold
+from sklearn.linear_model import Ridge
+from sklearn.model_selection import cross_val_score
+from sklearn.preprocessing import PolynomialFeatures
+
+# A seed just to ensure that the random numbers are the same for every run.
+np.random.seed(3155)
+# Generate the data.
+n = 100
+x = np.linspace(-3, 3, n).reshape(-1, 1)
+y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
+# Decide degree on polynomial to fit
+poly = PolynomialFeatures(degree = 10)
+
+# Decide which values of lambda to use
+nlambdas = 500
+lambdas = np.logspace(-3, 5, nlambdas)
+# Initialize a KFold instance
+k = 5
+kfold = KFold(n_splits = k)
+estimated_mse_sklearn = np.zeros(nlambdas)
+i = 0
+for lmb in lambdas:
+    ridge = Ridge(alpha = lmb)
+    estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
+    estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
+    i += 1
+plt.figure()
+plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
+plt.xlabel('log10(lambda)')
+plt.ylabel('MSE')
+plt.legend()
+plt.show()
+
+
+
+
+
+
+ + + + +
+ + + + +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/chapter3.html b/doc/LectureNotes/_build/html/chapter3.html new file mode 100644 index 000000000..dd2bd0aa6 --- /dev/null +++ b/doc/LectureNotes/_build/html/chapter3.html @@ -0,0 +1,1102 @@ + + + + + + + + 3. Ridge and Lasso Regression — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ + +
+
+ +
+ +
+

3. Ridge and Lasso Regression

+

Video of Lecture

+
+

3.1. The singular value decomposition

+

The examples we have looked at so far are cases where we normally can +invert the matrix \(\boldsymbol{X}^T\boldsymbol{X}\). Using a polynomial expansion as we +did both for the masses and the fitting of the equation of state, +leads to row vectors of the design matrix which are essentially +orthogonal due to the polynomial character of our model. Obtaining the inverse of the design matrix is then often done via a so-called LU, QR or Cholesky decomposition.

+

This may +however not the be case in general and a standard matrix inversion +algorithm based on say LU, QR or Cholesky decomposition may lead to singularities. We will see examples of this below.

+

There is however a way to partially circumvent this problem and also gain some insights about the ordinary least squares approach, and later shrinkage methods like Ridge and Lasso regressions.

+

This is given by the Singular Value Decomposition algorithm, perhaps +the most powerful linear algebra algorithm. Let us look at a +different example where we may have problems with the standard matrix +inversion algorithm. Thereafter we dive into the math of the SVD.

+

One of the typical problems we encounter with linear regression, in particular +when the matrix \(\boldsymbol{X}\) (our so-called design matrix) is high-dimensional, +are problems with near singular or singular matrices. The column vectors of \(\boldsymbol{X}\) +may be linearly dependent, normally referred to as super-collinearity.
+This means that the matrix may be rank deficient and it is basically impossible to +to model the data using linear regression. As an example, consider the matrix

+
+\[\begin{split} +\begin{align*} +\mathbf{X} & = \left[ +\begin{array}{rrr} +1 & -1 & 2 +\\ +1 & 0 & 1 +\\ +1 & 2 & -1 +\\ +1 & 1 & 0 +\end{array} \right] +\end{align*} +\end{split}\]
+

The columns of \(\boldsymbol{X}\) are linearly dependent. We see this easily since the +the first column is the row-wise sum of the other two columns. The rank (more correct, +the column rank) of a matrix is the dimension of the space spanned by the +column vectors. Hence, the rank of \(\mathbf{X}\) is equal to the number +of linearly independent columns. In this particular case the matrix has rank 2.

+

Super-collinearity of an \((n \times p)\)-dimensional design matrix \(\mathbf{X}\) implies +that the inverse of the matrix \(\boldsymbol{X}^T\boldsymbol{X}\) (the matrix we need to invert to solve the linear regression equations) is non-invertible. If we have a square matrix that does not have an inverse, we say this matrix singular. The example here demonstrates this

+
+\[\begin{split} +\begin{align*} +\boldsymbol{X} & = \left[ +\begin{array}{rr} +1 & -1 +\\ +1 & -1 +\end{array} \right]. +\end{align*} +\end{split}\]
+

We see easily that \(\mbox{det}(\boldsymbol{X}) = x_{11} x_{22} - x_{12} x_{21} = 1 \times (-1) - 1 \times (-1) = 0\). Hence, \(\mathbf{X}\) is singular and its inverse is undefined. +This is equivalent to saying that the matrix \(\boldsymbol{X}\) has at least an eigenvalue which is zero.

+

If our design matrix \(\boldsymbol{X}\) which enters the linear regression problem

+ +
+
+\[ +\begin{equation} +\boldsymbol{\beta} = (\boldsymbol{X}^{T} \boldsymbol{X})^{-1} \boldsymbol{X}^{T} \boldsymbol{y}, +\label{_auto1} \tag{1} +\end{equation} +\]
+

has linearly dependent column vectors, we will not be able to compute the inverse +of \(\boldsymbol{X}^T\boldsymbol{X}\) and we cannot find the parameters (estimators) \(\beta_i\). +The estimators are only well-defined if \((\boldsymbol{X}^{T}\boldsymbol{X})^{-1}\) exits. +This is more likely to happen when the matrix \(\boldsymbol{X}\) is high-dimensional. In this case it is likely to encounter a situation where +the regression parameters \(\beta_i\) cannot be estimated.

+

A cheap ad hoc approach is simply to add a small diagonal component to the matrix to invert, that is we change

+
+\[ +\boldsymbol{X}^{T} \boldsymbol{X} \rightarrow \boldsymbol{X}^{T} \boldsymbol{X}+\lambda \boldsymbol{I}, +\]
+

where \(\boldsymbol{I}\) is the identity matrix. When we discuss Ridge regression this is actually what we end up evaluating. The parameter \(\lambda\) is called a hyperparameter. More about this later.

+

From standard linear algebra we know that a square matrix \(\boldsymbol{X}\) can be diagonalized if and only it is +a so-called normal matrix, that is if \(\boldsymbol{X}\in {\mathbb{R}}^{n\times n}\) +we have \(\boldsymbol{X}\boldsymbol{X}^T=\boldsymbol{X}^T\boldsymbol{X}\) or if \(\boldsymbol{X}\in {\mathbb{C}}^{n\times n}\) we have \(\boldsymbol{X}\boldsymbol{X}^{\dagger}=\boldsymbol{X}^{\dagger}\boldsymbol{X}\). +The matrix has then a set of eigenpairs

+
+\[ +(\lambda_1,\boldsymbol{u}_1),\dots, (\lambda_n,\boldsymbol{u}_n), +\]
+

and the eigenvalues are given by the diagonal matrix

+
+\[ +\boldsymbol{\Sigma}=\mathrm{Diag}(\lambda_1, \dots,\lambda_n). +\]
+

The matrix \(\boldsymbol{X}\) can be written in terms of an orthogonal/unitary transformation \(\boldsymbol{U}\)

+
+\[ +\boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T, +\]
+

with \(\boldsymbol{U}\boldsymbol{U}^T=\boldsymbol{I}\) or \(\boldsymbol{U}\boldsymbol{U}^{\dagger}=\boldsymbol{I}\).

+

Not all square matrices are diagonalizable. A matrix like the one discussed above

+
+\[\begin{split} +\boldsymbol{X} = \begin{bmatrix} +1& -1 \\ +1& -1\\ +\end{bmatrix} +\end{split}\]
+

is not diagonalizable, it is a so-called defective matrix. It is easy to see that the condition +\(\boldsymbol{X}\boldsymbol{X}^T=\boldsymbol{X}^T\boldsymbol{X}\) is not fulfilled.

+
+
+

3.2. The SVD, a Fantastic Algorithm

+

However, and this is the strength of the SVD algorithm, any general +matrix \(\boldsymbol{X}\) can be decomposed in terms of a diagonal matrix and +two orthogonal/unitary matrices. The Singular Value Decompostion +(SVD) theorem +states that a general \(m\times n\) matrix \(\boldsymbol{X}\) can be written in +terms of a diagonal matrix \(\boldsymbol{\Sigma}\) of dimensionality \(m\times n\) +and two orthognal matrices \(\boldsymbol{U}\) and \(\boldsymbol{V}\), where the first has +dimensionality \(m \times m\) and the last dimensionality \(n\times n\). +We have then

+
+\[ +\boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T +\]
+

As an example, the above defective matrix can be decomposed as

+
+\[\begin{split} +\boldsymbol{X} = \frac{1}{\sqrt{2}}\begin{bmatrix} 1& 1 \\ 1& -1\\ \end{bmatrix} \begin{bmatrix} 2& 0 \\ 0& 0\\ \end{bmatrix} \frac{1}{\sqrt{2}}\begin{bmatrix} 1& -1 \\ 1& 1\\ \end{bmatrix}=\boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T, +\end{split}\]
+

with eigenvalues \(\sigma_1=2\) and \(\sigma_2=0\). +The SVD exits always!

+

The SVD +decomposition (singular values) gives eigenvalues +\(\sigma_i\geq\sigma_{i+1}\) for all \(i\) and for dimensions larger than \(i=p\), the +eigenvalues (singular values) are zero.

+

In the general case, where our design matrix \(\boldsymbol{X}\) has dimension +\(n\times p\), the matrix is thus decomposed into an \(n\times n\) +orthogonal matrix \(\boldsymbol{U}\), a \(p\times p\) orthogonal matrix \(\boldsymbol{V}\) +and a diagonal matrix \(\boldsymbol{\Sigma}\) with \(r=\mathrm{min}(n,p)\) +singular values \(\sigma_i\geq 0\) on the main diagonal and zeros filling +the rest of the matrix. There are at most \(p\) singular values +assuming that \(n > p\). In our regression examples for the nuclear +masses and the equation of state this is indeed the case, while for +the Ising model we have \(p > n\). These are often cases that lead to +near singular or singular matrices.

+

The columns of \(\boldsymbol{U}\) are called the left singular vectors while the columns of \(\boldsymbol{V}\) are the right singular vectors.

+
+
+

3.3. Economy-size SVD

+

If we assume that \(n > p\), then our matrix \(\boldsymbol{U}\) has dimension \(n +\times n\). The last \(n-p\) columns of \(\boldsymbol{U}\) become however +irrelevant in our calculations since they are multiplied with the +zeros in \(\boldsymbol{\Sigma}\).

+

The economy-size decomposition removes extra rows or columns of zeros +from the diagonal matrix of singular values, \(\boldsymbol{\Sigma}\), along with the columns +in either \(\boldsymbol{U}\) or \(\boldsymbol{V}\) that multiply those zeros in the expression. +Removing these zeros and columns can improve execution time +and reduce storage requirements without compromising the accuracy of +the decomposition.

+

If \(n > p\), we keep only the first \(p\) columns of \(\boldsymbol{U}\) and \(\boldsymbol{\Sigma}\) has dimension \(p\times p\). +If \(p > n\), then only the first \(n\) columns of \(\boldsymbol{V}\) are computed and \(\boldsymbol{\Sigma}\) has dimension \(n\times n\). +The \(n=p\) case is obvious, we retain the full SVD. +In general the economy-size SVD leads to less FLOPS and still conserving the desired accuracy.

+
+
+
import numpy as np
+# SVD inversion
+def SVDinv(A):
+    ''' Takes as input a numpy matrix A and returns inv(A) based on singular value decomposition (SVD).
+    SVD is numerically more stable than the inversion algorithms provided by
+    numpy and scipy.linalg at the cost of being slower.
+    '''
+    U, s, VT = np.linalg.svd(A)
+#    print('test U')
+#    print( (np.transpose(U) @ U - U @np.transpose(U)))
+#    print('test VT')
+#    print( (np.transpose(VT) @ VT - VT @np.transpose(VT)))
+    print(U)
+    print(s)
+    print(VT)
+
+    D = np.zeros((len(U),len(VT)))
+    for i in range(0,len(VT)):
+        D[i,i]=s[i]
+    UT = np.transpose(U); V = np.transpose(VT); invD = np.linalg.inv(D)
+    return np.matmul(V,np.matmul(invD,UT))
+
+
+X = np.array([ [1.0, -1.0, 2.0], [1.0, 0.0, 1.0], [1.0, 2.0, -1.0], [1.0, 1.0, 0.0] ])
+print(X)
+A = np.transpose(X) @ X
+print(A)
+# Brute force inversion of super-collinear matrix
+#B = np.linalg.inv(A)
+#print(B)
+C = SVDinv(A)
+print(C)
+
+
+
+
+
[[ 1. -1.  2.]
+ [ 1.  0.  1.]
+ [ 1.  2. -1.]
+ [ 1.  1.  0.]]
+[[ 4.  2.  2.]
+ [ 2.  6. -4.]
+ [ 2. -4.  6.]]
+[[-1.18404906e-16  8.16496581e-01 -5.77350269e-01]
+ [-7.07106781e-01  4.08248290e-01  5.77350269e-01]
+ [ 7.07106781e-01  4.08248290e-01  5.77350269e-01]]
+[1.00000000e+01 6.00000000e+00 9.10898112e-32]
+[[ 3.33066907e-17 -7.07106781e-01  7.07106781e-01]
+ [ 8.16496581e-01  4.08248290e-01  4.08248290e-01]
+ [ 5.77350269e-01 -5.77350269e-01 -5.77350269e-01]]
+[[-3.65939208e+30  3.65939208e+30  3.65939208e+30]
+ [ 3.65939208e+30 -3.65939208e+30 -3.65939208e+30]
+ [ 3.65939208e+30 -3.65939208e+30 -3.65939208e+30]]
+
+
+
+
+

The matrix \(\boldsymbol{X}\) has columns that are linearly dependent. The first +column is the row-wise sum of the other two columns. The rank of a +matrix (the column rank) is the dimension of space spanned by the +column vectors. The rank of the matrix is the number of linearly +independent columns, in this case just \(2\). We see this from the +singular values when running the above code. Running the standard +inversion algorithm for matrix inversion with \(\boldsymbol{X}^T\boldsymbol{X}\) results +in the program terminating due to a singular matrix.

+

There are several interesting mathematical properties which will be +relevant when we are going to discuss the differences between say +ordinary least squares (OLS) and Ridge regression.

+

We have from OLS that the parameters of the linear approximation are given by

+
+\[ +\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\beta} = \boldsymbol{X}\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}. +\]
+

The matrix to invert can be rewritten in terms of our SVD decomposition as

+
+\[ +\boldsymbol{X}^T\boldsymbol{X} = \boldsymbol{V}\boldsymbol{\Sigma}^T\boldsymbol{U}^T\boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T. +\]
+

Using the orthogonality properties of \(\boldsymbol{U}\) we have

+
+\[ +\boldsymbol{X}^T\boldsymbol{X} = \boldsymbol{V}\boldsymbol{\Sigma}^T\boldsymbol{\Sigma}\boldsymbol{V}^T = \boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T, +\]
+

with \(\boldsymbol{D}\) being a diagonal matrix with values along the diagonal given by the singular values squared.

+

This means that

+
+\[ +(\boldsymbol{X}^T\boldsymbol{X})\boldsymbol{V} = \boldsymbol{V}\boldsymbol{D}, +\]
+

that is the eigenvectors of \((\boldsymbol{X}^T\boldsymbol{X})\) are given by the columns of the right singular matrix of \(\boldsymbol{X}\) and the eigenvalues are the squared singular values. It is easy to show (show this) that

+
+\[ +(\boldsymbol{X}\boldsymbol{X}^T)\boldsymbol{U} = \boldsymbol{U}\boldsymbol{D}, +\]
+

that is, the eigenvectors of \((\boldsymbol{X}\boldsymbol{X})^T\) are the columns of the left singular matrix and the eigenvalues are the same.

+

Going back to our OLS equation we have

+
+\[ +\boldsymbol{X}\boldsymbol{\beta} = \boldsymbol{X}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T \right)^{-1}\boldsymbol{X}^T\boldsymbol{y}=\boldsymbol{U\Sigma V^T}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T \right)^{-1}(\boldsymbol{U\Sigma V^T})^T\boldsymbol{y}=\boldsymbol{U}\boldsymbol{U}^T\boldsymbol{y}. +\]
+

We will come back to this expression when we discuss Ridge regression.

+

$\( \tilde{y}^{OLS}=\boldsymbol{X}\hat{\beta}^{OLS}=\sum_{j=1}^p \boldsymbol{u}_j\boldsymbol{u}_j^T\boldsymbol{y}\)$ and for Ridge we have

+

$\( \tilde{y}^{Ridge}=\boldsymbol{X}\hat{\beta}^{Ridge}=\sum_{j=1}^p \boldsymbol{u}_j\frac{\sigma_j^2}{\sigma_j^2+\lambda}\boldsymbol{u}_j^T\boldsymbol{y}\)$ .

+

It is indeed the economy-sized SVD, note the summation runs up tp $\(p\)\( only and not \)\(n\)$.

+

Here we have that $\(\boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T\)\(, with \)\(\Sigma\)\( being an \)\( n\times p\)\( matrix and \)\(\boldsymbol{V}\)\( being a \)\( p\times p\)\( matrix. We also have assumed here that \)\( n > p\)$.

+
+
+

3.4. Ridge and LASSO Regression

+

Video of Lecture

+

Let us remind ourselves about the expression for the standard Mean Squared Error (MSE) which we used to define our cost function and the equations for the ordinary least squares (OLS) method, that is +our optimization problem is

+
+\[ +{\displaystyle \min_{\boldsymbol{\beta}\in {\mathbb{R}}^{p}}}\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. +\]
+

or we can state it as

+
+\[ +{\displaystyle \min_{\boldsymbol{\beta}\in +{\mathbb{R}}^{p}}}\frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2=\frac{1}{n}\vert\vert \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\vert\vert_2^2, +\]
+

where we have used the definition of a norm-2 vector, that is

+
+\[ +\vert\vert \boldsymbol{x}\vert\vert_2 = \sqrt{\sum_i x_i^2}. +\]
+

By minimizing the above equation with respect to the parameters +\(\boldsymbol{\beta}\) we could then obtain an analytical expression for the +parameters \(\boldsymbol{\beta}\). We can add a regularization parameter \(\lambda\) by +defining a new cost function to be optimized, that is

+
+\[ +{\displaystyle \min_{\boldsymbol{\beta}\in +{\mathbb{R}}^{p}}}\frac{1}{n}\vert\vert \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\vert\vert_2^2+\lambda\vert\vert \boldsymbol{\beta}\vert\vert_2^2 +\]
+

which leads to the Ridge regression minimization problem where we +require that \(\vert\vert \boldsymbol{\beta}\vert\vert_2^2\le t\), where \(t\) is +a finite number larger than zero. By defining

+
+\[ +C(\boldsymbol{X},\boldsymbol{\beta})=\frac{1}{n}\vert\vert \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\vert\vert_2^2+\lambda\vert\vert \boldsymbol{\beta}\vert\vert_1, +\]
+

we have a new optimization equation

+
+\[ +{\displaystyle \min_{\boldsymbol{\beta}\in +{\mathbb{R}}^{p}}}\frac{1}{n}\vert\vert \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\vert\vert_2^2+\lambda\vert\vert \boldsymbol{\beta}\vert\vert_1 +\]
+

which leads to Lasso regression. Lasso stands for least absolute shrinkage and selection operator.

+

Here we have defined the norm-1 as

+
+\[ +\vert\vert \boldsymbol{x}\vert\vert_1 = \sum_i \vert x_i\vert. +\]
+

Using the matrix-vector expression for Ridge regression,

+
+\[ +C(\boldsymbol{X},\boldsymbol{\beta})=\frac{1}{n}\left\{(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta})^T(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta})\right\}+\lambda\boldsymbol{\beta}^T\boldsymbol{\beta}, +\]
+

by taking the derivatives with respect to \(\boldsymbol{\beta}\) we obtain then +a slightly modified matrix inversion problem which for finite values +of \(\lambda\) does not suffer from singularity problems. We obtain

+
+\[ +\boldsymbol{\beta}^{\mathrm{Ridge}} = \left(\boldsymbol{X}^T\boldsymbol{X}+\lambda\boldsymbol{I}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}, +\]
+

with \(\boldsymbol{I}\) being a \(p\times p\) identity matrix with the constraint that

+
+\[ +\sum_{i=0}^{p-1} \beta_i^2 \leq t, +\]
+

with \(t\) a finite positive number.

+

We see that Ridge regression is nothing but the standard +OLS with a modified diagonal term added to \(\boldsymbol{X}^T\boldsymbol{X}\). The +consequences, in particular for our discussion of the bias-variance tradeoff +are rather interesting.

+

Furthermore, if we use the result above in terms of the SVD decomposition (our analysis was done for the OLS method), we had

+
+\[ +(\boldsymbol{X}\boldsymbol{X}^T)\boldsymbol{U} = \boldsymbol{U}\boldsymbol{D}. +\]
+

We can analyse the OLS solutions in terms of the eigenvectors (the columns) of the right singular value matrix \(\boldsymbol{U}\) as

+
+\[ +\boldsymbol{X}\boldsymbol{\beta} = \boldsymbol{X}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T \right)^{-1}\boldsymbol{X}^T\boldsymbol{y}=\boldsymbol{U\Sigma V^T}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T \right)^{-1}(\boldsymbol{U\Sigma V^T})^T\boldsymbol{y}=\boldsymbol{U}\boldsymbol{U}^T\boldsymbol{y} +\]
+

For Ridge regression this becomes

+
+\[ +\boldsymbol{X}\boldsymbol{\beta}^{\mathrm{Ridge}} = \boldsymbol{U\Sigma V^T}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T+\lambda\boldsymbol{I} \right)^{-1}(\boldsymbol{U\Sigma V^T})^T\boldsymbol{y}=\sum_{j=0}^{p-1}\boldsymbol{u}_j\boldsymbol{u}_j^T\frac{\sigma_j^2}{\sigma_j^2+\lambda}\boldsymbol{y}, +\]
+

with the vectors \(\boldsymbol{u}_j\) being the columns of \(\boldsymbol{U}\).

+

Since \(\lambda \geq 0\), it means that compared to OLS, we have

+
+\[ +\frac{\sigma_j^2}{\sigma_j^2+\lambda} \leq 1. +\]
+

Ridge regression finds the coordinates of \(\boldsymbol{y}\) with respect to the +orthonormal basis \(\boldsymbol{U}\), it then shrinks the coordinates by +\(\frac{\sigma_j^2}{\sigma_j^2+\lambda}\). Recall that the SVD has +eigenvalues ordered in a descending way, that is \(\sigma_i \geq +\sigma_{i+1}\).

+

For small eigenvalues \(\sigma_i\) it means that their contributions become less important, a fact which can be used to reduce the number of degrees of freedom. +Actually, calculating the variance of \(\boldsymbol{X}\boldsymbol{v}_j\) shows that this quantity is equal to \(\sigma_j^2/n\). +With a parameter \(\lambda\) we can thus shrink the role of specific parameters.

+

For the sake of simplicity, let us assume that the design matrix is orthonormal, that is

+
+\[ +\boldsymbol{X}^T\boldsymbol{X}=(\boldsymbol{X}^T\boldsymbol{X})^{-1} =\boldsymbol{I}. +\]
+

In this case the standard OLS results in

+
+\[ +\boldsymbol{\beta}^{\mathrm{OLS}} = \boldsymbol{X}^T\boldsymbol{y}=\sum_{i=0}^{p-1}\boldsymbol{u}_j\boldsymbol{u}_j^T\boldsymbol{y}, +\]
+

and

+
+\[ +\boldsymbol{\beta}^{\mathrm{Ridge}} = \left(\boldsymbol{I}+\lambda\boldsymbol{I}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}=\left(1+\lambda\right)^{-1}\boldsymbol{\beta}^{\mathrm{OLS}}, +\]
+

that is the Ridge estimator scales the OLS estimator by the inverse of a factor \(1+\lambda\), and +the Ridge estimator converges to zero when the hyperparameter goes to +infinity.

+

We will come back to more interpreations after we have gone through some of the statistical analysis part.

+

For more discussions of Ridge and Lasso regression, Wessel van Wieringen’s article is highly recommended. +Similarly, Mehta et al’s article is also recommended.

+
+
+

3.5. A better understanding of regularization

+

The parameter \(\lambda\) that we have introduced in the Ridge (and +Lasso as well) regression is often called a regularization parameter +or shrinkage parameter. It is common to call it a hyperparameter. What does it mean mathemtically?

+

Here we will first look at how to analyze the difference between the +standard OLS equations and the Ridge expressions in terms of a linear +algebra analysis using the SVD algorithm. Thereafter, we will link +(see the material on the bias-variance tradeoff below) these +observation to the statisical analysis of the results. In particular +we consider how the variance of the parameters \(\boldsymbol{\beta}\) is +affected by changing the parameter \(\lambda\).

+

We have our design matrix +\(\boldsymbol{X}\in {\mathbb{R}}^{n\times p}\). With the SVD we decompose it as

+
+\[ +\boldsymbol{X} = \boldsymbol{U\Sigma V^T}, +\]
+

with \(\boldsymbol{U}\in {\mathbb{R}}^{n\times n}\), \(\boldsymbol{\Sigma}\in {\mathbb{R}}^{n\times p}\) +and \(\boldsymbol{V}\in {\mathbb{R}}^{p\times p}\).

+

The matrices \(\boldsymbol{U}\) and \(\boldsymbol{V}\) are unitary/orthonormal matrices, that is in case the matrices are real we have \(\boldsymbol{U}^T\boldsymbol{U}=\boldsymbol{U}\boldsymbol{U}^T=\boldsymbol{I}\) and \(\boldsymbol{V}^T\boldsymbol{V}=\boldsymbol{V}\boldsymbol{V}^T=\boldsymbol{I}\).

+
+
+

3.6. Introducing the Covariance and Correlation functions

+

Before we discuss the link between for example Ridge regression and the singular value decomposition, we need to remind ourselves about +the definition of the covariance and the correlation function. These are quantities

+

Suppose we have defined two vectors +\(\hat{x}\) and \(\hat{y}\) with \(n\) elements each. The covariance matrix \(\boldsymbol{C}\) is defined as

+
+\[\begin{split} +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} \mathrm{cov}[\boldsymbol{x},\boldsymbol{x}] & \mathrm{cov}[\boldsymbol{x},\boldsymbol{y}] \\ + \mathrm{cov}[\boldsymbol{y},\boldsymbol{x}] & \mathrm{cov}[\boldsymbol{y},\boldsymbol{y}] \\ + \end{bmatrix}, +\end{split}\]
+

where for example

+
+\[ +\mathrm{cov}[\boldsymbol{x},\boldsymbol{y}] =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +\]
+

With this definition and recalling that the variance is defined as

+
+\[ +\mathrm{var}[\boldsymbol{x}]=\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})^2, +\]
+

we can rewrite the covariance matrix as

+
+\[\begin{split} +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} \mathrm{var}[\boldsymbol{x}] & \mathrm{cov}[\boldsymbol{x},\boldsymbol{y}] \\ + \mathrm{cov}[\boldsymbol{x},\boldsymbol{y}] & \mathrm{var}[\boldsymbol{y}] \\ + \end{bmatrix}. +\end{split}\]
+

The covariance takes values between zero and infinity and may thus +lead to problems with loss of numerical precision for particularly +large values. It is common to scale the covariance matrix by +introducing instead the correlation matrix defined via the so-called +correlation function

+
+\[ +\mathrm{corr}[\boldsymbol{x},\boldsymbol{y}]=\frac{\mathrm{cov}[\boldsymbol{x},\boldsymbol{y}]}{\sqrt{\mathrm{var}[\boldsymbol{x}] \mathrm{var}[\boldsymbol{y}]}}. +\]
+

The correlation function is then given by values \(\mathrm{corr}[\boldsymbol{x},\boldsymbol{y}] +\in [-1,1]\). This avoids eventual problems with too large values. We +can then define the correlation matrix for the two vectors \(\boldsymbol{x}\) +and \(\boldsymbol{y}\) as

+
+\[\begin{split} +\boldsymbol{K}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} 1 & \mathrm{corr}[\boldsymbol{x},\boldsymbol{y}] \\ + \mathrm{corr}[\boldsymbol{y},\boldsymbol{x}] & 1 \\ + \end{bmatrix}, +\end{split}\]
+

In the above example this is the function we constructed using pandas.

+

In our derivation of the various regression algorithms like Ordinary Least Squares or Ridge regression +we defined the design/feature matrix \(\boldsymbol{X}\) as

+
+\[\begin{split} +\boldsymbol{X}=\begin{bmatrix} +x_{0,0} & x_{0,1} & x_{0,2}& \dots & \dots x_{0,p-1}\\ +x_{1,0} & x_{1,1} & x_{1,2}& \dots & \dots x_{1,p-1}\\ +x_{2,0} & x_{2,1} & x_{2,2}& \dots & \dots x_{2,p-1}\\ +\dots & \dots & \dots & \dots \dots & \dots \\ +x_{n-2,0} & x_{n-2,1} & x_{n-2,2}& \dots & \dots x_{n-2,p-1}\\ +x_{n-1,0} & x_{n-1,1} & x_{n-1,2}& \dots & \dots x_{n-1,p-1}\\ +\end{bmatrix}, +\end{split}\]
+

with \(\boldsymbol{X}\in {\mathbb{R}}^{n\times p}\), with the predictors/features \(p\) refering to the column numbers and the +entries \(n\) being the row elements. +We can rewrite the design/feature matrix in terms of its column vectors as

+
+\[ +\boldsymbol{X}=\begin{bmatrix} \boldsymbol{x}_0 & \boldsymbol{x}_1 & \boldsymbol{x}_2 & \dots & \dots & \boldsymbol{x}_{p-1}\end{bmatrix}, +\]
+

with a given vector

+
+\[ +\boldsymbol{x}_i^T = \begin{bmatrix}x_{0,i} & x_{1,i} & x_{2,i}& \dots & \dots x_{n-1,i}\end{bmatrix}. +\]
+

With these definitions, we can now rewrite our \(2\times 2\) +correaltion/covariance matrix in terms of a moe general design/feature +matrix \(\boldsymbol{X}\in {\mathbb{R}}^{n\times p}\). This leads to a \(p\times p\) +covariance matrix for the vectors \(\boldsymbol{x}_i\) with \(i=0,1,\dots,p-1\)

+
+\[\begin{split} +\boldsymbol{C}[\boldsymbol{x}] = \begin{bmatrix} +\mathrm{var}[\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_1] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_2] & \dots & \dots & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_{p-1}]\\ +\mathrm{cov}[\boldsymbol{x}_1,\boldsymbol{x}_0] & \mathrm{var}[\boldsymbol{x}_1] & \mathrm{cov}[\boldsymbol{x}_1,\boldsymbol{x}_2] & \dots & \dots & \mathrm{cov}[\boldsymbol{x}_1,\boldsymbol{x}_{p-1}]\\ +\mathrm{cov}[\boldsymbol{x}_2,\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_2,\boldsymbol{x}_1] & \mathrm{var}[\boldsymbol{x}_2] & \dots & \dots & \mathrm{cov}[\boldsymbol{x}_2,\boldsymbol{x}_{p-1}]\\ +\dots & \dots & \dots & \dots & \dots & \dots \\ +\dots & \dots & \dots & \dots & \dots & \dots \\ +\mathrm{cov}[\boldsymbol{x}_{p-1},\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_{p-1},\boldsymbol{x}_1] & \mathrm{cov}[\boldsymbol{x}_{p-1},\boldsymbol{x}_{2}] & \dots & \dots & \mathrm{var}[\boldsymbol{x}_{p-1}]\\ +\end{bmatrix}, +\end{split}\]
+

and the correlation matrix

+
+\[\begin{split} +\boldsymbol{K}[\boldsymbol{x}] = \begin{bmatrix} +1 & \mathrm{corr}[\boldsymbol{x}_0,\boldsymbol{x}_1] & \mathrm{corr}[\boldsymbol{x}_0,\boldsymbol{x}_2] & \dots & \dots & \mathrm{corr}[\boldsymbol{x}_0,\boldsymbol{x}_{p-1}]\\ +\mathrm{corr}[\boldsymbol{x}_1,\boldsymbol{x}_0] & 1 & \mathrm{corr}[\boldsymbol{x}_1,\boldsymbol{x}_2] & \dots & \dots & \mathrm{corr}[\boldsymbol{x}_1,\boldsymbol{x}_{p-1}]\\ +\mathrm{corr}[\boldsymbol{x}_2,\boldsymbol{x}_0] & \mathrm{corr}[\boldsymbol{x}_2,\boldsymbol{x}_1] & 1 & \dots & \dots & \mathrm{corr}[\boldsymbol{x}_2,\boldsymbol{x}_{p-1}]\\ +\dots & \dots & \dots & \dots & \dots & \dots \\ +\dots & \dots & \dots & \dots & \dots & \dots \\ +\mathrm{corr}[\boldsymbol{x}_{p-1},\boldsymbol{x}_0] & \mathrm{corr}[\boldsymbol{x}_{p-1},\boldsymbol{x}_1] & \mathrm{corr}[\boldsymbol{x}_{p-1},\boldsymbol{x}_{2}] & \dots & \dots & 1\\ +\end{bmatrix}, +\end{split}\]
+

The Numpy function np.cov calculates the covariance elements using +the factor \(1/(n-1)\) instead of \(1/n\) since it assumes we do not have +the exact mean values. The following simple function uses the +np.vstack function which takes each vector of dimension \(1\times n\) +and produces a \(2\times n\) matrix \(\boldsymbol{W}\)

+
+\[\begin{split} +\boldsymbol{W} = \begin{bmatrix} x_0 & y_0 \\ + x_1 & y_1 \\ + x_2 & y_2\\ + \dots & \dots \\ + x_{n-2} & y_{n-2}\\ + x_{n-1} & y_{n-1} & + \end{bmatrix}, +\end{split}\]
+

which in turn is converted into into the \(2\times 2\) covariance matrix +\(\boldsymbol{C}\) via the Numpy function np.cov(). We note that we can also calculate +the mean value of each set of samples \(\boldsymbol{x}\) etc using the Numpy +function np.mean(x). We can also extract the eigenvalues of the +covariance matrix through the np.linalg.eig() function.

+
+
+
# Importing various packages
+import numpy as np
+n = 100
+x = np.random.normal(size=n)
+print(np.mean(x))
+y = 4+3*x+np.random.normal(size=n)
+print(np.mean(y))
+W = np.vstack((x, y))
+C = np.cov(W)
+print(C)
+
+
+
+
+
-0.03494744740413562
+3.896222705525891
+[[ 1.15493691  3.39431833]
+ [ 3.39431833 10.95648659]]
+
+
+
+
+

The previous example can be converted into the correlation matrix by +simply scaling the matrix elements with the variances. We should also +subtract the mean values for each column. This leads to the following +code which sets up the correlations matrix for the previous example in +a more brute force way. Here we scale the mean values for each column of the design matrix, calculate the relevant mean values and variances and then finally set up the \(2\times 2\) correlation matrix (since we have only two vectors).

+
+
+
import numpy as np
+n = 100
+# define two vectors                                                                                           
+x = np.random.random(size=n)
+y = 4+3*x+np.random.normal(size=n)
+#scaling the x and y vectors                                                                                   
+x = x - np.mean(x)
+y = y - np.mean(y)
+variance_x = np.sum(x@x)/n
+variance_y = np.sum(y@y)/n
+print(variance_x)
+print(variance_y)
+cov_xy = np.sum(x@y)/n
+cov_xx = np.sum(x@x)/n
+cov_yy = np.sum(y@y)/n
+C = np.zeros((2,2))
+C[0,0]= cov_xx/variance_x
+C[1,1]= cov_yy/variance_y
+C[0,1]= cov_xy/np.sqrt(variance_y*variance_x)
+C[1,0]= C[0,1]
+print(C)
+
+
+
+
+
0.0684659365902349
+1.3864659735909566
+[[1.         0.56083552]
+ [0.56083552 1.        ]]
+
+
+
+
+

We see that the matrix elements along the diagonal are one as they +should be and that the matrix is symmetric. Furthermore, diagonalizing +this matrix we easily see that it is a positive definite matrix.

+

The above procedure with numpy can be made more compact if we use pandas.

+

We whow here how we can set up the correlation matrix using pandas, as done in this simple code

+
+
+
import numpy as np
+import pandas as pd
+n = 10
+x = np.random.normal(size=n)
+x = x - np.mean(x)
+y = 4+3*x+np.random.normal(size=n)
+y = y - np.mean(y)
+X = (np.vstack((x, y))).T
+print(X)
+Xpd = pd.DataFrame(X)
+print(Xpd)
+correlation_matrix = Xpd.corr()
+print(correlation_matrix)
+
+
+
+
+
[[-1.0123897  -2.8039812 ]
+ [-0.28718135 -0.92440839]
+ [-0.74988099 -2.62660334]
+ [-0.10484789 -0.22717222]
+ [-0.62761155 -1.59313927]
+ [ 0.99624087  3.35932784]
+ [-0.26202974 -1.03435975]
+ [-0.36630178 -0.10312915]
+ [-0.4877836  -1.72320998]
+ [ 2.90178573  7.67667546]]
+          0         1
+0 -1.012390 -2.803981
+1 -0.287181 -0.924408
+2 -0.749881 -2.626603
+3 -0.104848 -0.227172
+4 -0.627612 -1.593139
+5  0.996241  3.359328
+6 -0.262030 -1.034360
+7 -0.366302 -0.103129
+8 -0.487784 -1.723210
+9  2.901786  7.676675
+          0         1
+0  1.000000  0.989696
+1  0.989696  1.000000
+
+
+
+
+

We expand this model to the Franke function discussed above.

+
+
+
# Common imports
+import numpy as np
+import pandas as pd
+
+
+def FrankeFunction(x,y):
+	term1 = 0.75*np.exp(-(0.25*(9*x-2)**2) - 0.25*((9*y-2)**2))
+	term2 = 0.75*np.exp(-((9*x+1)**2)/49.0 - 0.1*(9*y+1))
+	term3 = 0.5*np.exp(-(9*x-7)**2/4.0 - 0.25*((9*y-3)**2))
+	term4 = -0.2*np.exp(-(9*x-4)**2 - (9*y-7)**2)
+	return term1 + term2 + term3 + term4
+
+
+def create_X(x, y, n ):
+	if len(x.shape) > 1:
+		x = np.ravel(x)
+		y = np.ravel(y)
+
+	N = len(x)
+	l = int((n+1)*(n+2)/2)		# Number of elements in beta
+	X = np.ones((N,l))
+
+	for i in range(1,n+1):
+		q = int((i)*(i+1)/2)
+		for k in range(i+1):
+			X[:,q+k] = (x**(i-k))*(y**k)
+
+	return X
+
+
+# Making meshgrid of datapoints and compute Franke's function
+n = 4
+N = 100
+x = np.sort(np.random.uniform(0, 1, N))
+y = np.sort(np.random.uniform(0, 1, N))
+z = FrankeFunction(x, y)
+X = create_X(x, y, n=n)    
+
+Xpd = pd.DataFrame(X)
+# subtract the mean values and set up the covariance matrix
+Xpd = Xpd - Xpd.mean()
+covariance_matrix = Xpd.cov()
+print(covariance_matrix)
+
+
+
+
+
     0         1         2         3         4         5         6         7   \
+0   0.0  0.000000  0.000000  0.000000  0.000000  0.000000  0.000000  0.000000   
+1   0.0  0.075348  0.082602  0.073564  0.078039  0.082574  0.065469  0.068545   
+2   0.0  0.082602  0.093536  0.081952  0.088312  0.094670  0.072999  0.077152   
+3   0.0  0.073564  0.081952  0.077804  0.082702  0.087489  0.072629  0.075932   
+4   0.0  0.078039  0.088312  0.082702  0.088620  0.094412  0.076979  0.080887   
+5   0.0  0.082574  0.094670  0.087489  0.094412  0.101207  0.081135  0.085647   
+6   0.0  0.065469  0.072999  0.072629  0.076979  0.081135  0.069875  0.072824   
+7   0.0  0.068545  0.077152  0.075932  0.080887  0.085647  0.072824  0.076146   
+8   0.0  0.071758  0.081453  0.079293  0.084869  0.090255  0.075775  0.079482   
+9   0.0  0.075149  0.085972  0.082762  0.088986  0.095035  0.078769  0.082882   
+10  0.0  0.057678  0.063987  0.065995  0.069640  0.073058  0.064814  0.067321   
+11  0.0  0.059991  0.066970  0.068481  0.072514  0.076323  0.067071  0.069824   
+12  0.0  0.062435  0.070113  0.071062  0.075503  0.079726  0.069384  0.072399   
+13  0.0  0.065032  0.073451  0.073761  0.078639  0.083307  0.071774  0.075069   
+14  0.0  0.067805  0.077022  0.076602  0.081951  0.087104  0.074261  0.077858   
+
+          8         9         10        11        12        13        14  
+0   0.000000  0.000000  0.000000  0.000000  0.000000  0.000000  0.000000  
+1   0.071758  0.075149  0.057678  0.059991  0.062435  0.065032  0.067805  
+2   0.081453  0.085972  0.063987  0.066970  0.070113  0.073451  0.077022  
+3   0.079293  0.082762  0.065995  0.068481  0.071062  0.073761  0.076602  
+4   0.084869  0.088986  0.069640  0.072514  0.075503  0.078639  0.081951  
+5   0.090255  0.095035  0.073058  0.076323  0.079726  0.083307  0.087104  
+6   0.075775  0.078769  0.064814  0.067071  0.069384  0.071774  0.074261  
+7   0.079482  0.082882  0.067321  0.069824  0.072399  0.075069  0.077858  
+8   0.083219  0.087047  0.069793  0.072554  0.075401  0.078366  0.081476  
+9   0.087047  0.091332  0.072268  0.075299  0.078437  0.081717  0.085171  
+10  0.069793  0.072268  0.061016  0.062978  0.064967  0.067002  0.069095  
+11  0.072554  0.075299  0.062978  0.065108  0.067277  0.069504  0.071804  
+12  0.075401  0.078437  0.064967  0.067277  0.069637  0.072069  0.074593  
+13  0.078366  0.081717  0.067002  0.069504  0.072069  0.074724  0.077490  
+14  0.081476  0.085171  0.069095  0.071804  0.074593  0.077490  0.080521  
+
+
+
+
+

We note here that the covariance is zero for the first rows and +columns since all matrix elements in the design matrix were set to one +(we are fitting the function in terms of a polynomial of degree \(n\)).

+

This means that the variance for these elements will be zero and will +cause problems when we set up the correlation matrix. We can simply +drop these elements and construct a correlation +matrix without these elements.

+

We can rewrite the covariance matrix in a more compact form in terms of the design/feature matrix \(\boldsymbol{X}\) as

+
+\[ +\boldsymbol{C}[\boldsymbol{x}] = \frac{1}{n}\boldsymbol{X}^T\boldsymbol{X}= \mathbb{E}[\boldsymbol{X}^T\boldsymbol{X}]. +\]
+

To see this let us simply look at a design matrix \(\boldsymbol{X}\in {\mathbb{R}}^{2\times 2}\)

+
+\[\begin{split} +\boldsymbol{X}=\begin{bmatrix} +x_{00} & x_{01}\\ +x_{10} & x_{11}\\ +\end{bmatrix}=\begin{bmatrix} +\boldsymbol{x}_{0} & \boldsymbol{x}_{1}\\ +\end{bmatrix}. +\end{split}\]
+

If we then compute the expectation value

+
+\[\begin{split} +\mathbb{E}[\boldsymbol{X}^T\boldsymbol{X}] = \frac{1}{n}\boldsymbol{X}^T\boldsymbol{X}=\begin{bmatrix} +x_{00}^2+x_{01}^2 & x_{00}x_{10}+x_{01}x_{11}\\ +x_{10}x_{00}+x_{11}x_{01} & x_{10}^2+x_{11}^2\\ +\end{bmatrix}, +\end{split}\]
+

which is just

+
+\[\begin{split} +\boldsymbol{C}[\boldsymbol{x}_0,\boldsymbol{x}_1] = \boldsymbol{C}[\boldsymbol{x}]=\begin{bmatrix} \mathrm{var}[\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_1] \\ + \mathrm{cov}[\boldsymbol{x}_1,\boldsymbol{x}_0] & \mathrm{var}[\boldsymbol{x}_1] \\ + \end{bmatrix}, +\end{split}\]
+

where we wrote $\(\boldsymbol{C}[\boldsymbol{x}_0,\boldsymbol{x}_1] = \boldsymbol{C}[\boldsymbol{x}]\)\( to indicate that this the covariance of the vectors \)\boldsymbol{x}\( of the design/feature matrix \)\boldsymbol{X}$.

+

It is easy to generalize this to a matrix \(\boldsymbol{X}\in {\mathbb{R}}^{n\times p}\).

+
+
+

3.7. Linking with SVD

+
+
+ + + + +
+ + + + +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/chapter4.html b/doc/LectureNotes/_build/html/chapter4.html new file mode 100644 index 000000000..4eb75a803 --- /dev/null +++ b/doc/LectureNotes/_build/html/chapter4.html @@ -0,0 +1,2217 @@ + + + + + + + + 4. Logistic Regression — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ +
+
+ +
+ + + + + + + + + + + + + + +
+ + +
+ +
+ + Contents +
+ + +
+
+
+
+
+ +
+ +
+

4. Logistic Regression

+

Video of Lecture

+
+

4.1. Logistic Regression

+

In linear regression our main interest was centered on learning the +coefficients of a functional fit (say a polynomial) in order to be +able to predict the response of a continuous variable on some unseen +data. The fit to the continuous variable \(y_i\) is based on some +independent variables \(\hat{x}_i\). Linear regression resulted in +analytical expressions for standard ordinary Least Squares or Ridge +regression (in terms of matrices to invert) for several quantities, +ranging from the variance and thereby the confidence intervals of the +parameters \(\hat{\beta}\) to the mean squared error. If we can invert +the product of the design matrices, linear regression gives then a +simple recipe for fitting our data.

+

Classification problems, however, are concerned with outcomes taking +the form of discrete variables (i.e. categories). We may for example, +on the basis of DNA sequencing for a number of patients, like to find +out which mutations are important for a certain disease; or based on +scans of various patients’ brains, figure out if there is a tumor or +not; or given a specific physical system, we’d like to identify its +state, say whether it is an ordered or disordered system (typical +situation in solid state physics); or classify the status of a +patient, whether she/he has a stroke or not and many other similar +situations.

+

The most common situation we encounter when we apply logistic +regression is that of two possible outcomes, normally denoted as a +binary outcome, true or false, positive or negative, success or +failure etc.

+

Logistic regression will also serve as our stepping stone towards +neural network algorithms and supervised deep learning. For logistic +learning, the minimization of the cost function leads to a non-linear +equation in the parameters \(\hat{\beta}\). The optimization of the +problem calls therefore for minimization algorithms. This forms the +bottle neck of all machine learning algorithms, namely how to find +reliable minima of a multi-variable function. This leads us to the +family of gradient descent methods. The latter are the working horses +of basically all modern machine learning algorithms.

+

We note also that many of the topics discussed here on logistic +regression are also commonly used in modern supervised Deep Learning +models, as we will see later.

+
+
+

4.2. Basics

+

We consider the case where the dependent variables, also called the +responses or the outcomes, \(y_i\) are discrete and only take values +from \(k=0,\dots,K-1\) (i.e. \(K\) classes).

+

The goal is to predict the +output classes from the design matrix \(\hat{X}\in\mathbb{R}^{n\times p}\) +made of \(n\) samples, each of which carries \(p\) features or predictors. The +primary goal is to identify the classes to which new unseen samples +belong.

+

Let us specialize to the case of two classes only, with outputs +\(y_i=0\) and \(y_i=1\). Our outcomes could represent the status of a +credit card user that could default or not on her/his credit card +debt. That is

+
+\[\begin{split} +y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}. +\end{split}\]
+

Before moving to the logistic model, let us try to use our linear +regression model to classify these two outcomes. We could for example +fit a linear model to the default case if \(y_i > 0.5\) and the no +default case \(y_i \leq 0.5\).

+

We would then have our +weighted linear combination, namely

+ +
+
+\[ +\begin{equation} +\hat{y} = \hat{X}^T\hat{\beta} + \hat{\epsilon}, +\label{_auto1} \tag{1} +\end{equation} +\]
+

where \(\hat{y}\) is a vector representing the possible outcomes, \(\hat{X}\) is our +\(n\times p\) design matrix and \(\hat{\beta}\) represents our estimators/predictors.

+

The main problem with our function is that it takes values on the +entire real axis. In the case of logistic regression, however, the +labels \(y_i\) are discrete variables. A typical example is the credit +card data discussed below here, where we can set the state of +defaulting the debt to \(y_i=1\) and not to \(y_i=0\) for one the persons +in the data set (see the full example below).

+

One simple way to get a discrete output is to have sign +functions that map the output of a linear regressor to values \(\{0,1\}\), +\(f(s_i)=sign(s_i)=1\) if \(s_i\ge 0\) and 0 if otherwise. +We will encounter this model in our first demonstration of neural networks. Historically it is called the perceptron" model in the machine learning literature. This model is extremely simple. However, in many cases it is more favorable to use a soft” classifier that outputs +the probability of a given category. This leads us to the logistic function.

+

The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person’s against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful.

+
+
+
%matplotlib inline
+
+# Common imports
+import os
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from sklearn.linear_model import LinearRegression, Ridge, Lasso
+from sklearn.model_selection import train_test_split
+from sklearn.utils import resample
+from sklearn.metrics import mean_squared_error
+from IPython.display import display
+from pylab import plt, mpl
+plt.style.use('seaborn')
+mpl.rcParams['font.family'] = 'serif'
+
+# Where to save the figures and data files
+PROJECT_ROOT_DIR = "Results"
+FIGURE_ID = "Results/FigureFiles"
+DATA_ID = "DataFiles/"
+
+if not os.path.exists(PROJECT_ROOT_DIR):
+    os.mkdir(PROJECT_ROOT_DIR)
+
+if not os.path.exists(FIGURE_ID):
+    os.makedirs(FIGURE_ID)
+
+if not os.path.exists(DATA_ID):
+    os.makedirs(DATA_ID)
+
+def image_path(fig_id):
+    return os.path.join(FIGURE_ID, fig_id)
+
+def data_path(dat_id):
+    return os.path.join(DATA_ID, dat_id)
+
+def save_fig(fig_id):
+    plt.savefig(image_path(fig_id) + ".png", format='png')
+
+infile = open(data_path("chddata.csv"),'r')
+
+# Read the chd data as  csv file and organize the data into arrays with age group, age, and chd
+chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))
+chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']
+output = chd['CHD']
+age = chd['Age']
+agegroup = chd['Agegroup']
+numberID  = chd['ID'] 
+display(chd)
+
+plt.scatter(age, output, marker='o')
+plt.axis([18,70.0,-0.1, 1.2])
+plt.xlabel(r'Age')
+plt.ylabel(r'CHD')
+plt.title(r'Age distribution and Coronary heart disease')
+plt.show()
+
+
+
+
+
---------------------------------------------------------------------------
+FileNotFoundError                         Traceback (most recent call last)
+<ipython-input-1-a77d5ac269b2> in <module>
+     38     plt.savefig(image_path(fig_id) + ".png", format='png')
+     39 
+---> 40 infile = open(data_path("chddata.csv"),'r')
+     41 
+     42 # Read the chd data as  csv file and organize the data into arrays with age group, age, and chd
+
+FileNotFoundError: [Errno 2] No such file or directory: 'DataFiles/chddata.csv'
+
+
+
+
+

What we could attempt however is to plot the mean value for each group.

+
+
+
agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])
+group = np.array([1, 2, 3, 4, 5, 6, 7, 8])
+plt.plot(group, agegroupmean, "r-")
+plt.axis([0,9,0, 1.0])
+plt.xlabel(r'Age group')
+plt.ylabel(r'CHD mean values')
+plt.title(r'Mean values for each age group')
+plt.show()
+
+
+
+
+

We are now trying to find a function \(f(y\vert x)\), that is a function which gives us an expected value for the output \(y\) with a given input \(x\). +In standard linear regression with a linear dependence on \(x\), we would write this in terms of our model

+
+\[ +f(y_i\vert x_i)=\beta_0+\beta_1 x_i. +\]
+

This expression implies however that \(f(y_i\vert x_i)\) could take any +value from minus infinity to plus infinity. If we however let +\(f(y\vert y)\) be represented by the mean value, the above example +shows us that we can constrain the function to take values between +zero and one, that is we have \(0 \le f(y_i\vert x_i) \le 1\). Looking +at our last curve we see also that it has an S-shaped form. This leads +us to a very popular model for the function \(f\), namely the so-called +Sigmoid function or logistic model. We will consider this function as +representing the probability for finding a value of \(y_i\) with a given +\(x_i\).

+
+
+

4.3. The logistic function

+

Another widely studied model, is the so-called +perceptron model, which is an example of a “hard classification” model. We +will encounter this model when we discuss neural networks as +well. Each datapoint is deterministically assigned to a category (i.e +\(y_i=0\) or \(y_i=1\)). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a “soft” +classifier that outputs the probability of a given category rather +than a single value. For example, given \(x_i\), the classifier +outputs the probability of being in a category \(k\). Logistic regression +is the most common example of a so-called soft classifier. In logistic +regression, the probability that a data point \(x_i\) +belongs to a category \(y_i=\{0,1\}\) is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event,

+
+\[ +p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}. +\]
+

Note that \(1-p(t)= p(-t)\).

+
+
+

4.4. Examples of likelihood functions used in logistic regression and nueral networks

+

The following code plots the logistic function, the step function and other functions we will encounter from here and on.

+
+
+
"""The sigmoid function (or the logistic curve) is a
+function that takes any real number, z, and outputs a number (0,1).
+It is useful in neural networks for assigning weights on a relative scale.
+The value z is the weighted sum of parameters involved in the learning algorithm."""
+
+import numpy
+import matplotlib.pyplot as plt
+import math as mt
+
+z = numpy.arange(-5, 5, .1)
+sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))
+sigma = sigma_fn(z)
+
+fig = plt.figure()
+ax = fig.add_subplot(111)
+ax.plot(z, sigma)
+ax.set_ylim([-0.1, 1.1])
+ax.set_xlim([-5,5])
+ax.grid(True)
+ax.set_xlabel('z')
+ax.set_title('sigmoid function')
+
+plt.show()
+
+"""Step Function"""
+z = numpy.arange(-5, 5, .02)
+step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)
+step = step_fn(z)
+
+fig = plt.figure()
+ax = fig.add_subplot(111)
+ax.plot(z, step)
+ax.set_ylim([-0.5, 1.5])
+ax.set_xlim([-5,5])
+ax.grid(True)
+ax.set_xlabel('z')
+ax.set_title('step function')
+
+plt.show()
+
+"""tanh Function"""
+z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)
+t = numpy.tanh(z)
+
+fig = plt.figure()
+ax = fig.add_subplot(111)
+ax.plot(z, t)
+ax.set_ylim([-1.0, 1.0])
+ax.set_xlim([-2*mt.pi,2*mt.pi])
+ax.grid(True)
+ax.set_xlabel('z')
+ax.set_title('tanh function')
+
+plt.show()
+
+
+
+
+

We assume now that we have two classes with \(y_i\) either \(0\) or \(1\). Furthermore we assume also that we have only two parameters \(\beta\) in our fitting of the Sigmoid function, that is we define probabilities

+
+\[\begin{split} +\begin{align*} +p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\ +p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}), +\end{align*} +\end{split}\]
+

where \(\hat{\beta}\) are the weights we wish to extract from data, in our case \(\beta_0\) and \(\beta_1\).

+

Note that we used

+
+\[ +p(y_i=0\vert x_i, \hat{\beta}) = 1-p(y_i=1\vert x_i, \hat{\beta}). +\]
+

In order to define the total likelihood for all possible outcomes from a
+dataset \(\mathcal{D}=\{(y_i,x_i)\}\), with the binary labels +\(y_i\in\{0,1\}\) and where the data points are drawn independently, we use the so-called Maximum Likelihood Estimation (MLE) principle. +We aim thus at maximizing +the probability of seeing the observed data. We can then approximate the +likelihood in terms of the product of the individual probabilities of a specific outcome \(y_i\), that is

+
+\[\begin{split} +\begin{align*} +P(\mathcal{D}|\hat{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\hat{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\hat{\beta}))\right]^{1-y_i}\nonumber \\ +\end{align*} +\end{split}\]
+

from which we obtain the log-likelihood and our cost/loss function

+
+\[ +\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\hat{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\hat{\beta}))\right]\right). +\]
+

Reordering the logarithms, we can rewrite the cost/loss function as

+
+\[ +\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right). +\]
+

The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to \(\beta\). +Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that

+
+\[ +\mathcal{C}(\hat{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right). +\]
+

This equation is known in statistics as the cross entropy. Finally, we note that just as in linear regression, +in practice we often supplement the cross-entropy with additional regularization terms, usually \(L_1\) and \(L_2\) regularization as we did for Ridge and Lasso regression.

+

The cross entropy is a convex function of the weights \(\hat{\beta}\) and, +therefore, any local minimizer is a global minimizer.

+

Minimizing this +cost function with respect to the two parameters \(\beta_0\) and \(\beta_1\) we obtain

+
+\[ +\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right), +\]
+

and

+
+\[ +\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right). +\]
+

Let us now define a vector \(\hat{y}\) with \(n\) elements \(y_i\), an +\(n\times p\) matrix \(\hat{X}\) which contains the \(x_i\) values and a +vector \(\hat{p}\) of fitted probabilities \(p(y_i\vert x_i,\hat{\beta})\). We can rewrite in a more compact form the first +derivative of cost function as

+
+\[ +\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right). +\]
+

If we in addition define a diagonal matrix \(\hat{W}\) with elements +\(p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta})\), we can obtain a compact expression of the second derivative as

+
+\[ +\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}. +\]
+

Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with \(p\) predictors

+
+\[ +\log{ \frac{p(\hat{\beta}\hat{x})}{1-p(\hat{\beta}\hat{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p. +\]
+

Here we defined \(\hat{x}=[1,x_1,x_2,\dots,x_p]\) and \(\hat{\beta}=[\beta_0, \beta_1, \dots, \beta_p]\) leading to

+
+\[ +p(\hat{\beta}\hat{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}. +\]
+

Till now we have mainly focused on two classes, the so-called binary +system. Suppose we wish to extend to \(K\) classes. Let us for the sake +of simplicity assume we have only two predictors. We have then following model

+
+\[ +\log{\frac{p(C=1\vert x)}{p(K\vert x)}} = \beta_{10}+\beta_{11}x_1, +\]
+

and

+
+\[ +\log{\frac{p(C=2\vert x)}{p(K\vert x)}} = \beta_{20}+\beta_{21}x_1, +\]
+

and so on till the class \(C=K-1\) class

+
+\[ +\log{\frac{p(C=K-1\vert x)}{p(K\vert x)}} = \beta_{(K-1)0}+\beta_{(K-1)1}x_1, +\]
+

and the model is specified in term of \(K-1\) so-called log-odds or +logit transformations.

+

In our discussion of neural networks we will encounter the above again +in terms of a slightly modified function, the so-called Softmax function.

+

The softmax function is used in various multiclass classification +methods, such as multinomial logistic regression (also known as +softmax regression), multiclass linear discriminant analysis, naive +Bayes classifiers, and artificial neural networks. Specifically, in +multinomial logistic regression and linear discriminant analysis, the +input to the function is the result of \(K\) distinct linear functions, +and the predicted probability for the \(k\)-th class given a sample +vector \(\hat{x}\) and a weighting vector \(\hat{\beta}\) is (with two +predictors):

+
+\[ +p(C=k\vert \mathbf {x} )=\frac{\exp{(\beta_{k0}+\beta_{k1}x_1)}}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}}. +\]
+

It is easy to extend to more predictors. The final class is

+
+\[ +p(C=K\vert \mathbf {x} )=\frac{1}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}}, +\]
+

and they sum to one. Our earlier discussions were all specialized to +the case with two classes only. It is easy to see from the above that +what we derived earlier is compatible with these equations.

+

To find the optimal parameters we would typically use a gradient +descent method. Newton’s method and gradient descent methods are +discussed in the material on optimization +methods.

+
+
+

4.5. Wisconsin Cancer Data

+

We show here how we can use a simple regression case on the breast +cancer data using Logistic regression as our algorithm for +classification.

+
+
+
import matplotlib.pyplot as plt
+import numpy as np
+from sklearn.model_selection import  train_test_split 
+from sklearn.datasets import load_breast_cancer
+from sklearn.linear_model import LogisticRegression
+
+# Load the data
+cancer = load_breast_cancer()
+
+X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
+print(X_train.shape)
+print(X_test.shape)
+# Logistic Regression
+logreg = LogisticRegression(solver='lbfgs')
+logreg.fit(X_train, y_train)
+print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
+#now scale the data
+from sklearn.preprocessing import StandardScaler
+scaler = StandardScaler()
+scaler.fit(X_train)
+X_train_scaled = scaler.transform(X_train)
+X_test_scaled = scaler.transform(X_test)
+# Logistic Regression
+logreg.fit(X_train_scaled, y_train)
+print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
+
+
+
+
+

In addition to the above scores, we could also study the covariance (and the correlation matrix). +We use Pandas to compute the correlation matrix.

+
+
+
import matplotlib.pyplot as plt
+import numpy as np
+from sklearn.model_selection import  train_test_split 
+from sklearn.datasets import load_breast_cancer
+from sklearn.linear_model import LogisticRegression
+cancer = load_breast_cancer()
+import pandas as pd
+# Making a data frame
+cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
+
+fig, axes = plt.subplots(15,2,figsize=(10,20))
+malignant = cancer.data[cancer.target == 0]
+benign = cancer.data[cancer.target == 1]
+ax = axes.ravel()
+
+for i in range(30):
+    _, bins = np.histogram(cancer.data[:,i], bins =50)
+    ax[i].hist(malignant[:,i], bins = bins, alpha = 0.5)
+    ax[i].hist(benign[:,i], bins = bins, alpha = 0.5)
+    ax[i].set_title(cancer.feature_names[i])
+    ax[i].set_yticks(())
+ax[0].set_xlabel("Feature magnitude")
+ax[0].set_ylabel("Frequency")
+ax[0].legend(["Malignant", "Benign"], loc ="best")
+fig.tight_layout()
+plt.show()
+
+import seaborn as sns
+correlation_matrix = cancerpd.corr().round(1)
+# use the heatmap function from seaborn to plot the correlation matrix
+# annot = True to print the values inside the square
+plt.figure(figsize=(15,8))
+sns.heatmap(data=correlation_matrix, annot=True)
+plt.show()
+
+
+
+
+

In the above example we note two things. In the first plot we display +the overlap of benign and malignant tumors as functions of the various +features in the Wisconsing breast cancer data set. We see that for +some of the features we can distinguish clearly the benign and +malignant cases while for other features we cannot. This can point to +us which features may be of greater interest when we wish to classify +a benign or not benign tumour.

+

In the second figure we have computed the so-called correlation +matrix, which in our case with thirty features becomes a \(30\times 30\) +matrix.

+

We constructed this matrix using pandas via the statements

+
+
+
cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)
+
+
+
+
+

and then

+
+
+
correlation_matrix = cancerpd.corr().round(1)
+
+
+
+
+

Diagonalizing this matrix we can in turn say something about which +features are of relevance and which are not. This leads us to +the classical Principal Component Analysis (PCA) theorem with +applications. This will be discussed later this semester (week 43).

+
+
+
import matplotlib.pyplot as plt
+import numpy as np
+from sklearn.model_selection import  train_test_split 
+from sklearn.datasets import load_breast_cancer
+from sklearn.linear_model import LogisticRegression
+
+# Load the data
+cancer = load_breast_cancer()
+
+X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)
+print(X_train.shape)
+print(X_test.shape)
+# Logistic Regression
+logreg = LogisticRegression(solver='lbfgs')
+logreg.fit(X_train, y_train)
+print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test)))
+#now scale the data
+from sklearn.preprocessing import StandardScaler
+scaler = StandardScaler()
+scaler.fit(X_train)
+X_train_scaled = scaler.transform(X_train)
+X_test_scaled = scaler.transform(X_test)
+# Logistic Regression
+logreg.fit(X_train_scaled, y_train)
+print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
+
+
+from sklearn.preprocessing import LabelEncoder
+from sklearn.model_selection import cross_validate
+#Cross validation
+accuracy = cross_validate(logreg,X_test_scaled,y_test,cv=10)['test_score']
+print(accuracy)
+print("Test set accuracy with Logistic Regression  and scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test)))
+
+
+import scikitplot as skplt
+y_pred = logreg.predict(X_test_scaled)
+skplt.metrics.plot_confusion_matrix(y_test, y_pred, normalize=True)
+plt.show()
+y_probas = logreg.predict_proba(X_test_scaled)
+skplt.metrics.plot_roc(y_test, y_probas)
+plt.show()
+skplt.metrics.plot_cumulative_gain(y_test, y_probas)
+plt.show()
+
+
+
+
+
+
+

4.6. Optimization, the central part of any Machine Learning algortithm

+

Almost every problem in machine learning and data science starts with +a dataset \(X\), a model \(g(\beta)\), which is a function of the +parameters \(\beta\) and a cost function \(C(X, g(\beta))\) that allows +us to judge how well the model \(g(\beta)\) explains the observations +\(X\). The model is fit by finding the values of \(\beta\) that minimize +the cost function. Ideally we would be able to solve for \(\beta\) +analytically, however this is not possible in general and we must use +some approximative/numerical method to compute the minimum.

+
+
+

4.7. Revisiting our Logistic Regression case

+

In our discussion on Logistic Regression we studied the +case of +two classes, with \(y_i\) either +\(0\) or \(1\). Furthermore we assumed also that we have only two +parameters \(\beta\) in our fitting, that is we +defined probabilities

+
+\[\begin{split} +\begin{align*} +p(y_i=1|x_i,\boldsymbol{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\ +p(y_i=0|x_i,\boldsymbol{\beta}) &= 1 - p(y_i=1|x_i,\boldsymbol{\beta}), +\end{align*} +\end{split}\]
+

where \(\boldsymbol{\beta}\) are the weights we wish to extract from data, in our case \(\beta_0\) and \(\beta_1\).

+
+
+

4.8. The equations to solve

+

Our compact equations used a definition of a vector \(\boldsymbol{y}\) with \(n\) +elements \(y_i\), an \(n\times p\) matrix \(\boldsymbol{X}\) which contains the +\(x_i\) values and a vector \(\boldsymbol{p}\) of fitted probabilities +\(p(y_i\vert x_i,\boldsymbol{\beta})\). We rewrote in a more compact form +the first derivative of the cost function as

+
+\[ +\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{p}\right). +\]
+

If we in addition define a diagonal matrix \(\boldsymbol{W}\) with elements +\(p(y_i\vert x_i,\boldsymbol{\beta})(1-p(y_i\vert x_i,\boldsymbol{\beta})\), we can obtain a compact expression of the second derivative as

+
+\[ +\frac{\partial^2 \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}\partial \boldsymbol{\beta}^T} = \boldsymbol{X}^T\boldsymbol{W}\boldsymbol{X}. +\]
+

This defines what is called the Hessian matrix.

+
+
+

4.9. Solving using Newton-Raphson’s method

+

If we can set up these equations, Newton-Raphson’s iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives.

+

Our iterative scheme is then given by

+
+\[ +\boldsymbol{\beta}^{\mathrm{new}} = \boldsymbol{\beta}^{\mathrm{old}}-\left(\frac{\partial^2 \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}\partial \boldsymbol{\beta}^T}\right)^{-1}_{\boldsymbol{\beta}^{\mathrm{old}}}\times \left(\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}}\right)_{\boldsymbol{\beta}^{\mathrm{old}}}, +\]
+

or in matrix form as

+
+\[ +\boldsymbol{\beta}^{\mathrm{new}} = \boldsymbol{\beta}^{\mathrm{old}}-\left(\boldsymbol{X}^T\boldsymbol{W}\boldsymbol{X} \right)^{-1}\times \left(-\boldsymbol{X}^T(\boldsymbol{y}-\boldsymbol{p}) \right)_{\boldsymbol{\beta}^{\mathrm{old}}}. +\]
+

The right-hand side is computed with the old values of \(\beta\).

+

If we can compute these matrices, in particular the Hessian, the above is often the easiest method to implement.

+
+
+

4.10. Brief reminder on Newton-Raphson’s method

+

Let us quickly remind ourselves how we derive the above method.

+

Perhaps the most celebrated of all one-dimensional root-finding +routines is Newton’s method, also called the Newton-Raphson +method. This method requires the evaluation of both the +function \(f\) and its derivative \(f'\) at arbitrary points. +If you can only calculate the derivative +numerically and/or your function is not of the smooth type, we +normally discourage the use of this method.

+
+
+

4.11. The equations

+

The Newton-Raphson formula consists geometrically of extending the +tangent line at a current point until it crosses zero, then setting +the next guess to the abscissa of that zero-crossing. The mathematics +behind this method is rather simple. Employing a Taylor expansion for +\(x\) sufficiently close to the solution \(s\), we have

+ +
+
+\[ +f(s)=0=f(x)+(s-x)f'(x)+\frac{(s-x)^2}{2}f''(x) +\dots. + \label{eq:taylornr} \tag{2} +\]
+

For small enough values of the function and for well-behaved +functions, the terms beyond linear are unimportant, hence we obtain

+
+\[ +f(x)+(s-x)f'(x)\approx 0, +\]
+

yielding

+
+\[ +s\approx x-\frac{f(x)}{f'(x)}. +\]
+

Having in mind an iterative procedure, it is natural to start iterating with

+
+\[ +x_{n+1}=x_n-\frac{f(x_n)}{f'(x_n)}. +\]
+
+
+

4.12. Simple geometric interpretation

+

The above is Newton-Raphson’s method. It has a simple geometric +interpretation, namely \(x_{n+1}\) is the point where the tangent from +\((x_n,f(x_n))\) crosses the \(x\)-axis. Close to the solution, +Newton-Raphson converges fast to the desired result. However, if we +are far from a root, where the higher-order terms in the series are +important, the Newton-Raphson formula can give grossly inaccurate +results. For instance, the initial guess for the root might be so far +from the true root as to let the search interval include a local +maximum or minimum of the function. If an iteration places a trial +guess near such a local extremum, so that the first derivative nearly +vanishes, then Newton-Raphson may fail totally

+
+
+

4.13. Extending to more than one variable

+

Newton’s method can be generalized to systems of several non-linear equations +and variables. Consider the case with two equations

+
+\[\begin{split} +\begin{array}{cc} f_1(x_1,x_2) &=0\\ + f_2(x_1,x_2) &=0,\end{array} +\end{split}\]
+

which we Taylor expand to obtain

+
+\[\begin{split} +\begin{array}{cc} 0=f_1(x_1+h_1,x_2+h_2)=&f_1(x_1,x_2)+h_1 + \partial f_1/\partial x_1+h_2 + \partial f_1/\partial x_2+\dots\\ + 0=f_2(x_1+h_1,x_2+h_2)=&f_2(x_1,x_2)+h_1 + \partial f_2/\partial x_1+h_2 + \partial f_2/\partial x_2+\dots + \end{array}. +\end{split}\]
+

Defining the Jacobian matrix \({\bf \boldsymbol{J}}\) we have

+
+\[\begin{split} +{\bf \boldsymbol{J}}=\left( \begin{array}{cc} + \partial f_1/\partial x_1 & \partial f_1/\partial x_2 \\ + \partial f_2/\partial x_1 &\partial f_2/\partial x_2 + \end{array} \right), +\end{split}\]
+

we can rephrase Newton’s method as

+
+\[\begin{split} +\left(\begin{array}{c} x_1^{n+1} \\ x_2^{n+1} \end{array} \right)= +\left(\begin{array}{c} x_1^{n} \\ x_2^{n} \end{array} \right)+ +\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right), +\end{split}\]
+

where we have defined

+
+\[\begin{split} +\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right)= + -{\bf \boldsymbol{J}}^{-1} + \left(\begin{array}{c} f_1(x_1^{n},x_2^{n}) \\ f_2(x_1^{n},x_2^{n}) \end{array} \right). +\end{split}\]
+

We need thus to compute the inverse of the Jacobian matrix and it +is to understand that difficulties may +arise in case \({\bf \boldsymbol{J}}\) is nearly singular.

+

It is rather straightforward to extend the above scheme to systems of +more than two non-linear equations. In our case, the Jacobian matrix is given by the Hessian that represents the second derivative of cost function.

+
+
+

4.14. Steepest descent

+

The basic idea of gradient descent is +that a function \(F(\mathbf{x})\), +\(\mathbf{x} \equiv (x_1,\cdots,x_n)\), decreases fastest if one goes from \(\bf {x}\) in the +direction of the negative gradient \(-\nabla F(\mathbf{x})\).

+

It can be shown that if

+
+\[ +\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), +\]
+

with \(\gamma_k > 0\).

+

For \(\gamma_k\) small enough, then \(F(\mathbf{x}_{k+1}) \leq +F(\mathbf{x}_k)\). This means that for a sufficiently small \(\gamma_k\) +we are always moving towards smaller function values, i.e a minimum.

+
+
+

4.15. More on Steepest descent

+

The previous observation is the basis of the method of steepest +descent, which is also referred to as just gradient descent (GD). One +starts with an initial guess \(\mathbf{x}_0\) for a minimum of \(F\) and +computes new approximations according to

+
+\[ +\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), \ \ k \geq 0. +\]
+

The parameter \(\gamma_k\) is often referred to as the step length or +the learning rate within the context of Machine Learning.

+
+
+

4.16. The ideal

+

Ideally the sequence \(\{\mathbf{x}_k \}_{k=0}\) converges to a global +minimum of the function \(F\). In general we do not know if we are in a +global or local minimum. In the special case when \(F\) is a convex +function, all local minima are also global minima, so in this case +gradient descent can converge to the global solution. The advantage of +this scheme is that it is conceptually simple and straightforward to +implement. However the method in this form has some severe +limitations:

+

In machine learing we are often faced with non-convex high dimensional +cost functions with many local minima. Since GD is deterministic we +will get stuck in a local minimum, if the method converges, unless we +have a very good intial guess. This also implies that the scheme is +sensitive to the chosen initial condition.

+

Note that the gradient is a function of \(\mathbf{x} = +(x_1,\cdots,x_n)\) which makes it expensive to compute numerically.

+
+
+

4.17. The sensitiveness of the gradient descent

+

The gradient descent method +is sensitive to the choice of learning rate \(\gamma_k\). This is due +to the fact that we are only guaranteed that \(F(\mathbf{x}_{k+1}) \leq +F(\mathbf{x}_k)\) for sufficiently small \(\gamma_k\). The problem is to +determine an optimal learning rate. If the learning rate is chosen too +small the method will take a long time to converge and if it is too +large we can experience erratic behavior.

+

Many of these shortcomings can be alleviated by introducing +randomness. One such method is that of Stochastic Gradient Descent +(SGD), see below.

+
+
+

4.18. Convex functions

+

Ideally we want our cost/loss function to be convex(concave).

+

First we give the definition of a convex set: A set \(C\) in +\(\mathbb{R}^n\) is said to be convex if, for all \(x\) and \(y\) in \(C\) and +all \(t \in (0,1)\) , the point \((1 − t)x + ty\) also belongs to +C. Geometrically this means that every point on the line segment +connecting \(x\) and \(y\) is in \(C\) as discussed below.

+

The convex subsets of \(\mathbb{R}\) are the intervals of +\(\mathbb{R}\). Examples of convex sets of \(\mathbb{R}^2\) are the +regular polygons (triangles, rectangles, pentagons, etc…).

+
+
+

4.19. Convex function

+

Convex function: Let \(X \subset \mathbb{R}^n\) be a convex set. Assume that the function \(f: X \rightarrow \mathbb{R}\) is continuous, then \(f\) is said to be convex if $\(f(tx_1 + (1-t)x_2) \leq tf(x_1) + (1-t)f(x_2) \)\( for all \)x_1, x_2 \in X\( and for all \)t \in [0,1]\(. If \)\leq\( is replaced with a strict inequaltiy in the definition, we demand \)x_1 \neq x_2\( and \)t\in(0,1)\( then \)f\( is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting \)f(x_1)\( and \)f(x_2)\(, the value of the function on the interval \)[x_1,x_2]$ is always below the line as illustrated below.

+
+
+

4.20. Conditions on convex functions

+

In the following we state first and second-order conditions which +ensures convexity of a function \(f\). We write \(D_f\) to denote the +domain of \(f\), i.e the subset of \(R^n\) where \(f\) is defined. For more +details and proofs we refer to: [S. Boyd and L. Vandenberghe. Convex Optimization. Cambridge University Press](http://stanford.edu/boyd/cvxbook/, 2004).

+

First order condition.

+

Suppose \(f\) is differentiable (i.e \(\nabla f(x)\) is well defined for +all \(x\) in the domain of \(f\)). Then \(f\) is convex if and only if \(D_f\) +is a convex set and $\(f(y) \geq f(x) + \nabla f(x)^T (y-x) \)\( holds +for all \)x,y \in D_f\(. This condition means that for a convex function +the first order Taylor expansion (right hand side above) at any point +a global under estimator of the function. To convince yourself you can +make a drawing of \)f(x) = x^2+1\( and draw the tangent line to \)f(x)$ and +note that it is always below the graph.

+

Second order condition.

+

Assume that \(f\) is twice +differentiable, i.e the Hessian matrix exists at each point in +\(D_f\). Then \(f\) is convex if and only if \(D_f\) is a convex set and its +Hessian is positive semi-definite for all \(x\in D_f\). For a +single-variable function this reduces to \(f''(x) \geq 0\). Geometrically this means that \(f\) has nonnegative curvature +everywhere.

+

This condition is particularly useful since it gives us an procedure for determining if the function under consideration is convex, apart from using the definition.

+
+
+

4.21. More on convex functions

+

The next result is of great importance to us and the reason why we are +going on about convex functions. In machine learning we frequently +have to minimize a loss/cost function in order to find the best +parameters for the model we are considering.

+

Ideally we want the +global minimum (for high-dimensional models it is hard to know +if we have local or global minimum). However, if the cost/loss function +is convex the following result provides invaluable information:

+

Any minimum is global for convex functions.

+

Consider the problem of finding \(x \in \mathbb{R}^n\) such that \(f(x)\) +is minimal, where \(f\) is convex and differentiable. Then, any point +\(x^*\) that satisfies \(\nabla f(x^*) = 0\) is a global minimum.

+

This result means that if we know that the cost/loss function is convex and we are able to find a minimum, we are guaranteed that it is a global minimum.

+
+
+

4.22. Some simple problems

+
    +
  1. Show that \(f(x)=x^2\) is convex for \(x \in \mathbb{R}\) using the definition of convexity. Hint: If you re-write the definition, \(f\) is convex if the following holds for all \(x,y \in D_f\) and any \(\lambda \in [0,1]\) \(\lambda f(x)+(1-\lambda)f(y)-f(\lambda x + (1-\lambda) y ) \geq 0\).

  2. +
  3. Using the second order condition show that the following functions are convex on the specified domain.

  4. +
+
    +
  • \(f(x) = e^x\) is convex for \(x \in \mathbb{R}\).

  • +
  • \(g(x) = -\ln(x)\) is convex for \(x \in (0,\infty)\).

  • +
+
    +
  1. Let \(f(x) = x^2\) and \(g(x) = e^x\). Show that \(f(g(x))\) and \(g(f(x))\) is convex for \(x \in \mathbb{R}\). Also show that if \(f(x)\) is any convex function than \(h(x) = e^{f(x)}\) is convex.

  2. +
  3. A norm is any function that satisfy the following properties

  4. +
+
    +
  • \(f(\alpha x) = |\alpha| f(x)\) for all \(\alpha \in \mathbb{R}\).

  • +
  • \(f(x+y) \leq f(x) + f(y)\)

  • +
  • \(f(x) \leq 0\) for all \(x \in \mathbb{R}^n\) with equality if and only if \(x = 0\)

  • +
+

Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this).

+
+
+

4.23. Friday September 25

+

Video of Lecture and link to handwritten notes.

+
+
+

4.24. Standard steepest descent

+

Before we proceed, we would like to discuss the approach called the +standard Steepest descent (different from the above steepest descent discussion), which again leads to us having to be able +to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG).

+

The success of the CG method +for finding solutions of non-linear problems is based on the theory +of conjugate gradients for linear systems of equations. It belongs to +the class of iterative methods for solving problems from linear +algebra of the type

+
+\[ +\boldsymbol{A}\boldsymbol{x} = \boldsymbol{b}. +\]
+

In the iterative process we end up with a problem like

+
+\[ +\boldsymbol{r}= \boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}, +\]
+

where \(\boldsymbol{r}\) is the so-called residual or error in the iterative process.

+

When we have found the exact solution, \(\boldsymbol{r}=0\).

+
+
+

4.25. Gradient method

+

The residual is zero when we reach the minimum of the quadratic equation

+
+\[ +P(\boldsymbol{x})=\frac{1}{2}\boldsymbol{x}^T\boldsymbol{A}\boldsymbol{x} - \boldsymbol{x}^T\boldsymbol{b}, +\]
+

with the constraint that the matrix \(\boldsymbol{A}\) is positive definite and +symmetric. This defines also the Hessian and we want it to be positive definite.

+
+
+

4.26. Steepest descent method

+

We denote the initial guess for \(\boldsymbol{x}\) as \(\boldsymbol{x}_0\). +We can assume without loss of generality that

+
+\[ +\boldsymbol{x}_0=0, +\]
+

or consider the system

+
+\[ +\boldsymbol{A}\boldsymbol{z} = \boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_0, +\]
+

instead.

+
+
+

4.27. Steepest descent method

+

One can show that the solution \(\boldsymbol{x}\) is also the unique minimizer of the quadratic form

+
+\[ +f(\boldsymbol{x}) = \frac{1}{2}\boldsymbol{x}^T\boldsymbol{A}\boldsymbol{x} - \boldsymbol{x}^T \boldsymbol{x} , \quad \boldsymbol{x}\in\mathbf{R}^n. +\]
+

This suggests taking the first basis vector \(\boldsymbol{r}_1\) (see below for definition) +to be the gradient of \(f\) at \(\boldsymbol{x}=\boldsymbol{x}_0\), +which equals

+
+\[ +\boldsymbol{A}\boldsymbol{x}_0-\boldsymbol{b}, +\]
+

and +\(\boldsymbol{x}_0=0\) it is equal \(-\boldsymbol{b}\).

+
+
+

4.28. Final expressions

+

We can compute the residual iteratively as

+
+\[ +\boldsymbol{r}_{k+1}=\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_{k+1}, +\]
+

which equals

+
+\[ +\boldsymbol{b}-\boldsymbol{A}(\boldsymbol{x}_k+\alpha_k\boldsymbol{r}_k), +\]
+

or

+
+\[ +(\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_k)-\alpha_k\boldsymbol{A}\boldsymbol{r}_k, +\]
+

which gives

+
+\[ +\alpha_k = \frac{\boldsymbol{r}_k^T\boldsymbol{r}_k}{\boldsymbol{r}_k^T\boldsymbol{A}\boldsymbol{r}_k} +\]
+

leading to the iterative scheme

+
+\[ +\boldsymbol{x}_{k+1}=\boldsymbol{x}_k-\alpha_k\boldsymbol{r}_{k}, +\]
+
+
+

4.29. Steepest descent example

+
+
+
import numpy as np
+import numpy.linalg as la
+
+import scipy.optimize as sopt
+
+import matplotlib.pyplot as pt
+from mpl_toolkits.mplot3d import axes3d
+
+def f(x):
+    return 0.5*x[0]**2 + 2.5*x[1]**2
+
+def df(x):
+    return np.array([x[0], 5*x[1]])
+
+fig = pt.figure()
+ax = fig.gca(projection="3d")
+
+xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]
+fmesh = f(np.array([xmesh, ymesh]))
+ax.plot_surface(xmesh, ymesh, fmesh)
+
+
+
+
+

And then as countor plot

+
+
+
pt.axis("equal")
+pt.contour(xmesh, ymesh, fmesh)
+guesses = [np.array([2, 2./5])]
+
+
+
+
+

Find guesses

+
+
+
x = guesses[-1]
+s = -df(x)
+
+
+
+
+

Run it!

+
+
+
def f1d(alpha):
+    return f(x + alpha*s)
+
+alpha_opt = sopt.golden(f1d)
+next_guess = x + alpha_opt * s
+guesses.append(next_guess)
+print(next_guess)
+
+
+
+
+

What happened?

+
+
+
pt.axis("equal")
+pt.contour(xmesh, ymesh, fmesh, 50)
+it_array = np.array(guesses)
+pt.plot(it_array.T[0], it_array.T[1], "x-")
+
+
+
+
+
+
+

4.30. Conjugate gradient method

+

In the CG method we define so-called conjugate directions and two vectors +\(\boldsymbol{s}\) and \(\boldsymbol{t}\) +are said to be +conjugate if

+
+\[ +\boldsymbol{s}^T\boldsymbol{A}\boldsymbol{t}= 0. +\]
+

The philosophy of the CG method is to perform searches in various conjugate directions +of our vectors \(\boldsymbol{x}_i\) obeying the above criterion, namely

+
+\[ +\boldsymbol{x}_i^T\boldsymbol{A}\boldsymbol{x}_j= 0. +\]
+

Two vectors are conjugate if they are orthogonal with respect to +this inner product. Being conjugate is a symmetric relation: if \(\boldsymbol{s}\) is conjugate to \(\boldsymbol{t}\), then \(\boldsymbol{t}\) is conjugate to \(\boldsymbol{s}\).

+
+
+

4.31. Conjugate gradient method

+

An example is given by the eigenvectors of the matrix

+
+\[ +\boldsymbol{v}_i^T\boldsymbol{A}\boldsymbol{v}_j= \lambda\boldsymbol{v}_i^T\boldsymbol{v}_j, +\]
+

which is zero unless \(i=j\).

+
+
+

4.32. Conjugate gradient method

+

Assume now that we have a symmetric positive-definite matrix \(\boldsymbol{A}\) of size +\(n\times n\). At each iteration \(i+1\) we obtain the conjugate direction of a vector

+
+\[ +\boldsymbol{x}_{i+1}=\boldsymbol{x}_{i}+\alpha_i\boldsymbol{p}_{i}. +\]
+

We assume that \(\boldsymbol{p}_{i}\) is a sequence of \(n\) mutually conjugate directions. +Then the \(\boldsymbol{p}_{i}\) form a basis of \(R^n\) and we can expand the solution +\( \boldsymbol{A}\boldsymbol{x} = \boldsymbol{b}\) in this basis, namely

+
+\[ +\boldsymbol{x} = \sum^{n}_{i=1} \alpha_i \boldsymbol{p}_i. +\]
+
+
+

4.33. Conjugate gradient method

+

The coefficients are given by

+
+\[ +\mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. +\]
+

Multiplying with \(\boldsymbol{p}_k^T\) from the left gives

+
+\[ +\boldsymbol{p}_k^T \boldsymbol{A}\boldsymbol{x} = \sum^{n}_{i=1} \alpha_i\boldsymbol{p}_k^T \boldsymbol{A}\boldsymbol{p}_i= \boldsymbol{p}_k^T \boldsymbol{b}, +\]
+

and we can define the coefficients \(\alpha_k\) as

+
+\[ +\alpha_k = \frac{\boldsymbol{p}_k^T \boldsymbol{b}}{\boldsymbol{p}_k^T \boldsymbol{A} \boldsymbol{p}_k} +\]
+
+
+

4.34. Conjugate gradient method and iterations

+

If we choose the conjugate vectors \(\boldsymbol{p}_k\) carefully, +then we may not need all of them to obtain a good approximation to the solution +\(\boldsymbol{x}\). +We want to regard the conjugate gradient method as an iterative method. +This will us to solve systems where \(n\) is so large that the direct +method would take too much time.

+

We denote the initial guess for \(\boldsymbol{x}\) as \(\boldsymbol{x}_0\). +We can assume without loss of generality that

+
+\[ +\boldsymbol{x}_0=0, +\]
+

or consider the system

+
+\[ +\boldsymbol{A}\boldsymbol{z} = \boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_0, +\]
+

instead.

+
+
+

4.35. Conjugate gradient method

+

One can show that the solution \(\boldsymbol{x}\) is also the unique minimizer of the quadratic form

+
+\[ +f(\boldsymbol{x}) = \frac{1}{2}\boldsymbol{x}^T\boldsymbol{A}\boldsymbol{x} - \boldsymbol{x}^T \boldsymbol{x} , \quad \boldsymbol{x}\in\mathbf{R}^n. +\]
+

This suggests taking the first basis vector \(\boldsymbol{p}_1\) +to be the gradient of \(f\) at \(\boldsymbol{x}=\boldsymbol{x}_0\), +which equals

+
+\[ +\boldsymbol{A}\boldsymbol{x}_0-\boldsymbol{b}, +\]
+

and +\(\boldsymbol{x}_0=0\) it is equal \(-\boldsymbol{b}\). +The other vectors in the basis will be conjugate to the gradient, +hence the name conjugate gradient method.

+
+
+

4.36. Conjugate gradient method

+

Let \(\boldsymbol{r}_k\) be the residual at the \(k\)-th step:

+
+\[ +\boldsymbol{r}_k=\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_k. +\]
+

Note that \(\boldsymbol{r}_k\) is the negative gradient of \(f\) at +\(\boldsymbol{x}=\boldsymbol{x}_k\), +so the gradient descent method would be to move in the direction \(\boldsymbol{r}_k\). +Here, we insist that the directions \(\boldsymbol{p}_k\) are conjugate to each other, +so we take the direction closest to the gradient \(\boldsymbol{r}_k\)
+under the conjugacy constraint. +This gives the following expression

+
+\[ +\boldsymbol{p}_{k+1}=\boldsymbol{r}_k-\frac{\boldsymbol{p}_k^T \boldsymbol{A}\boldsymbol{r}_k}{\boldsymbol{p}_k^T\boldsymbol{A}\boldsymbol{p}_k} \boldsymbol{p}_k. +\]
+
+
+

4.37. Conjugate gradient method

+

We can also compute the residual iteratively as

+
+\[ +\boldsymbol{r}_{k+1}=\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_{k+1}, +\]
+

which equals

+
+\[ +\boldsymbol{b}-\boldsymbol{A}(\boldsymbol{x}_k+\alpha_k\boldsymbol{p}_k), +\]
+

or

+
+\[ +(\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_k)-\alpha_k\boldsymbol{A}\boldsymbol{p}_k, +\]
+

which gives

+
+\[ +\boldsymbol{r}_{k+1}=\boldsymbol{r}_k-\boldsymbol{A}\boldsymbol{p}_{k}, +\]
+
+
+

4.38. Revisiting our first homework

+

We will use linear regression as a case study for the gradient descent +methods. Linear regression is a great test case for the gradient +descent methods discussed in the lectures since it has several +desirable properties such as:

+
    +
  1. An analytical solution (recall homework set 1).

  2. +
  3. The gradient can be computed analytically.

  4. +
  5. The cost function is convex which guarantees that gradient descent converges for small enough learning rates

  6. +
+

We revisit an example similar to what we had in the first homework set. We had a function of the type

+
+
+
x = 2*np.random.rand(m,1)
+y = 4+3*x+np.random.randn(m,1)
+
+
+
+
+

with \(x_i \in [0,1] \) is chosen randomly using a uniform distribution. Additionally we have a stochastic noise chosen according to a normal distribution \(\cal {N}(0,1)\). +The linear regression model is given by

+
+\[ +h_\beta(x) = \boldsymbol{y} = \beta_0 + \beta_1 x, +\]
+

such that

+
+\[ +\boldsymbol{y}_i = \beta_0 + \beta_1 x_i. +\]
+
+
+

4.39. Gradient descent example

+

Let \(\mathbf{y} = (y_1,\cdots,y_n)^T\), \(\mathbf{\boldsymbol{y}} = (\boldsymbol{y}_1,\cdots,\boldsymbol{y}_n)^T\) and \(\beta = (\beta_0, \beta_1)^T\)

+

It is convenient to write \(\mathbf{\boldsymbol{y}} = X\beta\) where \(X \in \mathbb{R}^{100 \times 2} \) is the design matrix given by (we keep the intercept here)

+
+\[\begin{split} +X \equiv \begin{bmatrix} +1 & x_1 \\ +\vdots & \vdots \\ +1 & x_{100} & \\ +\end{bmatrix}. +\end{split}\]
+

The cost/loss/risk function is given by (

+
+\[ +C(\beta) = \frac{1}{n}||X\beta-\mathbf{y}||_{2}^{2} = \frac{1}{n}\sum_{i=1}^{100}\left[ (\beta_0 + \beta_1 x_i)^2 - 2 y_i (\beta_0 + \beta_1 x_i) + y_i^2\right] +\]
+

and we want to find \(\beta\) such that \(C(\beta)\) is minimized.

+
+
+

4.40. The derivative of the cost/loss function

+

Computing \(\partial C(\beta) / \partial \beta_0\) and \(\partial C(\beta) / \partial \beta_1\) we can show that the gradient can be written as

+
+\[\begin{split} +\nabla_{\beta} C(\beta) = \frac{2}{n}\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\ +\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\ +\end{bmatrix} = \frac{2}{n}X^T(X\beta - \mathbf{y}), +\end{split}\]
+

where \(X\) is the design matrix defined above.

+
+
+

4.41. The Hessian matrix

+

The Hessian matrix of \(C(\beta)\) is given by

+
+\[\begin{split} +\boldsymbol{H} \equiv \begin{bmatrix} +\frac{\partial^2 C(\beta)}{\partial \beta_0^2} & \frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} \\ +\frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} & \frac{\partial^2 C(\beta)}{\partial \beta_1^2} & \\ +\end{bmatrix} = \frac{2}{n}X^T X. +\end{split}\]
+

This result implies that \(C(\beta)\) is a convex function since the matrix \(X^T X\) always is positive semi-definite.

+
+
+

4.42. Simple program

+

We can now write a program that minimizes \(C(\beta)\) using the gradient descent method with a constant learning rate \(\gamma\) according to

+
+\[ +\beta_{k+1} = \beta_k - \gamma \nabla_\beta C(\beta_k), \ k=0,1,\cdots +\]
+

We can use the expression we computed for the gradient and let use a +\(\beta_0\) be chosen randomly and let \(\gamma = 0.001\). Stop iterating +when \(||\nabla_\beta C(\beta_k) || \leq \epsilon = 10^{-8}\). Note that the code below does not include the latter stop criterion.

+

And finally we can compare our solution for \(\beta\) with the analytic result given by +\(\beta= (X^TX)^{-1} X^T \mathbf{y}\).

+
+
+

4.43. Gradient Descent Example

+

Here our simple example

+
+
+
# Importing various packages
+from random import random, seed
+import numpy as np
+import matplotlib.pyplot as plt
+from mpl_toolkits.mplot3d import Axes3D
+from matplotlib import cm
+from matplotlib.ticker import LinearLocator, FormatStrFormatter
+import sys
+
+# the number of datapoints
+n = 100
+x = 2*np.random.rand(n,1)
+y = 4+3*x+np.random.randn(n,1)
+
+X = np.c_[np.ones((n,1)), x]
+# Hessian matrix
+H = (2.0/n)* X.T @ X
+# Get the eigenvalues
+EigValues, EigVectors = np.linalg.eig(H)
+print(EigValues)
+
+beta_linreg = np.linalg.inv(X.T @ X) @ X.T @ y
+print(beta_linreg)
+beta = np.random.randn(2,1)
+
+eta = 1.0/np.max(EigValues)
+Niterations = 1000
+
+for iter in range(Niterations):
+    gradient = (2.0/n)*X.T @ (X @ beta-y)
+    beta -= eta*gradient
+
+print(beta)
+xnew = np.array([[0],[2]])
+xbnew = np.c_[np.ones((2,1)), xnew]
+ypredict = xbnew.dot(beta)
+ypredict2 = xbnew.dot(beta_linreg)
+plt.plot(xnew, ypredict, "r-")
+plt.plot(xnew, ypredict2, "b-")
+plt.plot(x, y ,'ro')
+plt.axis([0,2.0,0, 15.0])
+plt.xlabel(r'$x$')
+plt.ylabel(r'$y$')
+plt.title(r'Gradient descent example')
+plt.show()
+
+
+
+
+
+
+

4.44. And a corresponding example using scikit-learn

+
+
+
# Importing various packages
+from random import random, seed
+import numpy as np
+import matplotlib.pyplot as plt
+from sklearn.linear_model import SGDRegressor
+
+n = 100
+x = 2*np.random.rand(n,1)
+y = 4+3*x+np.random.randn(n,1)
+
+X = np.c_[np.ones((n,1)), x]
+beta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)
+print(beta_linreg)
+sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1)
+sgdreg.fit(x,y.ravel())
+print(sgdreg.intercept_, sgdreg.coef_)
+
+
+
+
+
+
+

4.45. Gradient descent and Ridge

+

We have also discussed Ridge regression where the loss function contains a regularized term given by the \(L_2\) norm of \(\beta\),

+
+\[ +C_{\text{ridge}}(\beta) = \frac{1}{n}||X\beta -\mathbf{y}||^2 + \lambda ||\beta||^2, \ \lambda \geq 0. +\]
+

In order to minimize \(C_{\text{ridge}}(\beta)\) using GD we only have adjust the gradient as follows

+
+\[\begin{split} +\nabla_\beta C_{\text{ridge}}(\beta) = \frac{2}{n}\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\ +\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\ +\end{bmatrix} + 2\lambda\begin{bmatrix} \beta_0 \\ \beta_1\end{bmatrix} = 2 (X^T(X\beta - \mathbf{y})+\lambda \beta). +\end{split}\]
+

We can easily extend our program to minimize \(C_{\text{ridge}}(\beta)\) using gradient descent and compare with the analytical solution given by

+
+\[ +\beta_{\text{ridge}} = \left(X^T X + \lambda I_{2 \times 2} \right)^{-1} X^T \mathbf{y}. +\]
+
+
+

4.46. Program example for gradient descent with Ridge Regression

+
+
+
from random import random, seed
+import numpy as np
+import matplotlib.pyplot as plt
+from mpl_toolkits.mplot3d import Axes3D
+from matplotlib import cm
+from matplotlib.ticker import LinearLocator, FormatStrFormatter
+import sys
+
+# the number of datapoints
+n = 100
+x = 2*np.random.rand(n,1)
+y = 4+3*x+np.random.randn(n,1)
+
+X = np.c_[np.ones((n,1)), x]
+XT_X = X.T @ X
+
+#Ridge parameter lambda
+lmbda  = 0.001
+Id = lmbda* np.eye(XT_X.shape[0])
+
+beta_linreg = np.linalg.inv(XT_X+Id) @ X.T @ y
+print(beta_linreg)
+# Start plain gradient descent
+beta = np.random.randn(2,1)
+
+eta = 0.1
+Niterations = 100
+
+for iter in range(Niterations):
+    gradients = 2.0/n*X.T @ (X @ (beta)-y)+2*lmbda*beta
+    beta -= eta*gradients
+
+print(beta)
+ypredict = X @ beta
+ypredict2 = X @ beta_linreg
+plt.plot(x, ypredict, "r-")
+plt.plot(x, ypredict2, "b-")
+plt.plot(x, y ,'ro')
+plt.axis([0,2.0,0, 15.0])
+plt.xlabel(r'$x$')
+plt.ylabel(r'$y$')
+plt.title(r'Gradient descent example for Ridge')
+plt.show()
+
+
+
+
+
+
+

4.47. Using gradient descent methods, limitations

+
    +
  • Gradient descent (GD) finds local minima of our function. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our cost/loss/risk function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.

  • +
  • GD is sensitive to initial conditions. One consequence of the local nature of GD is that initial conditions matter. Depending on where one starts, one will end up at a different local minima. Therefore, it is very important to think about how one initializes the training process. This is true for GD as well as more complicated variants of GD.

  • +
  • Gradients are computationally expensive to calculate for large datasets. In many cases in statistics and ML, the cost/loss/risk function is a sum of terms, with one term for each data point. For example, in linear regression, \(E \propto \sum_{i=1}^n (y_i - \mathbf{w}^T\cdot\mathbf{x}_i)^2\); for logistic regression, the square error is replaced by the cross entropy. To calculate the gradient we have to sum over all \(n\) data points. Doing this at every GD step becomes extremely computationally expensive. An ingenious solution to this, is to calculate the gradients using small subsets of the data called “mini batches”. This has the added benefit of introducing stochasticity into our algorithm.

  • +
  • GD is very sensitive to choices of learning rates. GD is extremely sensitive to the choice of learning rates. If the learning rate is very small, the training process take an extremely long time. For larger learning rates, GD can diverge and give poor results. Furthermore, depending on what the local landscape looks like, we have to modify the learning rates to ensure convergence. Ideally, we would adaptively choose the learning rates to match the landscape.

  • +
  • GD treats all directions in parameter space uniformly. Another major drawback of GD is that unlike Newton’s method, the learning rate for GD is the same in all directions in parameter space. For this reason, the maximum learning rate is set by the behavior of the steepest direction and this can significantly slow down training. Ideally, we would like to take large steps in flat directions and small steps in steep directions. Since we are exploring rugged landscapes where curvatures change, this requires us to keep track of not only the gradient but second derivatives. The ideal scenario would be to calculate the Hessian but this proves to be too computationally expensive.

  • +
  • GD can take exponential time to escape saddle points, even with random initialization. As we mentioned, GD is extremely sensitive to initial condition since it determines the particular local minimum GD would eventually reach. However, even with a good initialization scheme, through the introduction of randomness, GD can still take exponential time to escape saddle points.

  • +
+
+
+

4.48. Stochastic Gradient Descent

+

Stochastic gradient descent (SGD) and variants thereof address some of +the shortcomings of the Gradient descent method discussed above.

+

The underlying idea of SGD comes from the observation that the cost +function, which we want to minimize, can almost always be written as a +sum over \(n\) data points \(\{\mathbf{x}_i\}_{i=1}^n\),

+
+\[ +C(\mathbf{\beta}) = \sum_{i=1}^n c_i(\mathbf{x}_i, +\mathbf{\beta}). +\]
+
+
+

4.49. Computation of gradients

+

This in turn means that the gradient can be +computed as a sum over \(i\)-gradients

+
+\[ +\nabla_\beta C(\mathbf{\beta}) = \sum_i^n \nabla_\beta c_i(\mathbf{x}_i, +\mathbf{\beta}). +\]
+

Stochasticity/randomness is introduced by only taking the +gradient on a subset of the data called minibatches. If there are \(n\) +data points and the size of each minibatch is \(M\), there will be \(n/M\) +minibatches. We denote these minibatches by \(B_k\) where +\(k=1,\cdots,n/M\).

+
+
+

4.50. SGD example

+

As an example, suppose we have \(10\) data points \((\mathbf{x}_1,\cdots, \mathbf{x}_{10})\) +and we choose to have \(M=5\) minibathces, +then each minibatch contains two data points. In particular we have +\(B_1 = (\mathbf{x}_1,\mathbf{x}_2), \cdots, B_5 = +(\mathbf{x}_9,\mathbf{x}_{10})\). Note that if you choose \(M=1\) you +have only a single batch with all data points and on the other extreme, +you may choose \(M=n\) resulting in a minibatch for each datapoint, i.e +\(B_k = \mathbf{x}_k\).

+

The idea is now to approximate the gradient by replacing the sum over +all data points with a sum over the data points in one the minibatches +picked at random in each gradient descent step

+
+\[ +\nabla_{\beta} +C(\mathbf{\beta}) = \sum_{i=1}^n \nabla_\beta c_i(\mathbf{x}_i, +\mathbf{\beta}) \rightarrow \sum_{i \in B_k}^n \nabla_\beta +c_i(\mathbf{x}_i, \mathbf{\beta}). +\]
+
+
+

4.51. The gradient step

+

Thus a gradient descent step now looks like

+
+\[ +\beta_{j+1} = \beta_j - \gamma_j \sum_{i \in B_k}^n \nabla_\beta c_i(\mathbf{x}_i, +\mathbf{\beta}) +\]
+

where \(k\) is picked at random with equal +probability from \([1,n/M]\). An iteration over the number of +minibathces (n/M) is commonly referred to as an epoch. Thus it is +typical to choose a number of epochs and for each epoch iterate over +the number of minibatches, as exemplified in the code below.

+
+
+

4.52. Simple example code

+
+
+
import numpy as np 
+
+n = 100 #100 datapoints 
+M = 5   #size of each minibatch
+m = int(n/M) #number of minibatches
+n_epochs = 10 #number of epochs
+
+j = 0
+for epoch in range(1,n_epochs+1):
+    for i in range(m):
+        k = np.random.randint(m) #Pick the k-th minibatch at random
+        #Compute the gradient using the data in minibatch Bk
+        #Compute new suggestion for 
+        j += 1
+
+
+
+
+

Taking the gradient only on a subset of the data has two important +benefits. First, it introduces randomness which decreases the chance +that our opmization scheme gets stuck in a local minima. Second, if +the size of the minibatches are small relative to the number of +datapoints (\(M < n\)), the computation of the gradient is much +cheaper since we sum over the datapoints in the \(k-th\) minibatch and not +all \(n\) datapoints.

+
+
+

4.53. When do we stop?

+

A natural question is when do we stop the search for a new minimum? +One possibility is to compute the full gradient after a given number +of epochs and check if the norm of the gradient is smaller than some +threshold and stop if true. However, the condition that the gradient +is zero is valid also for local minima, so this would only tell us +that we are close to a local/global minimum. However, we could also +evaluate the cost function at this point, store the result and +continue the search. If the test kicks in at a later stage we can +compare the values of the cost function and keep the \(\beta\) that +gave the lowest value.

+
+
+

4.54. Slightly different approach

+

Another approach is to let the step length \(\gamma_j\) depend on the +number of epochs in such a way that it becomes very small after a +reasonable time such that we do not move at all.

+

As an example, let \(e = 0,1,2,3,\cdots\) denote the current epoch and let \(t_0, t_1 > 0\) be two fixed numbers. Furthermore, let \(t = e \cdot m + i\) where \(m\) is the number of minibatches and \(i=0,\cdots,m-1\). Then the function $\(\gamma_j(t; t_0, t_1) = \frac{t_0}{t+t_1} \)\( goes to zero as the number of epochs gets large. I.e. we start with a step length \)\gamma_j (0; t_0, t_1) = t_0/t_1\( which decays in *time* \)t$.

+

In this way we can fix the number of epochs, compute \(\beta\) and +evaluate the cost function at the end. Repeating the computation will +give a different result since the scheme is random by design. Then we +pick the final \(\beta\) that gives the lowest value of the cost +function.

+
+
+
import numpy as np 
+
+def step_length(t,t0,t1):
+    return t0/(t+t1)
+
+n = 100 #100 datapoints 
+M = 5   #size of each minibatch
+m = int(n/M) #number of minibatches
+n_epochs = 500 #number of epochs
+t0 = 1.0
+t1 = 10
+
+gamma_j = t0/t1
+j = 0
+for epoch in range(1,n_epochs+1):
+    for i in range(m):
+        k = np.random.randint(m) #Pick the k-th minibatch at random
+        #Compute the gradient using the data in minibatch Bk
+        #Compute new suggestion for beta
+        t = epoch*m+i
+        gamma_j = step_length(t,t0,t1)
+        j += 1
+
+print("gamma_j after %d epochs: %g" % (n_epochs,gamma_j))
+
+
+
+
+
+
+

4.55. Program for stochastic gradient

+
+
+
# Importing various packages
+from math import exp, sqrt
+from random import random, seed
+import numpy as np
+import matplotlib.pyplot as plt
+from sklearn.linear_model import SGDRegressor
+
+m = 100
+x = 2*np.random.rand(m,1)
+y = 4+3*x+np.random.randn(m,1)
+
+X = np.c_[np.ones((m,1)), x]
+theta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)
+print("Own inversion")
+print(theta_linreg)
+sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1)
+sgdreg.fit(x,y.ravel())
+print("sgdreg from scikit")
+print(sgdreg.intercept_, sgdreg.coef_)
+
+
+theta = np.random.randn(2,1)
+eta = 0.1
+Niterations = 1000
+
+
+for iter in range(Niterations):
+    gradients = 2.0/m*X.T @ ((X @ theta)-y)
+    theta -= eta*gradients
+print("theta from own gd")
+print(theta)
+
+xnew = np.array([[0],[2]])
+Xnew = np.c_[np.ones((2,1)), xnew]
+ypredict = Xnew.dot(theta)
+ypredict2 = Xnew.dot(theta_linreg)
+
+
+n_epochs = 50
+t0, t1 = 5, 50
+def learning_schedule(t):
+    return t0/(t+t1)
+
+theta = np.random.randn(2,1)
+
+for epoch in range(n_epochs):
+    for i in range(m):
+        random_index = np.random.randint(m)
+        xi = X[random_index:random_index+1]
+        yi = y[random_index:random_index+1]
+        gradients = 2 * xi.T @ ((xi @ theta)-yi)
+        eta = learning_schedule(epoch*m+i)
+        theta = theta - eta*gradients
+print("theta from own sdg")
+print(theta)
+
+plt.plot(xnew, ypredict, "r-")
+plt.plot(xnew, ypredict2, "b-")
+plt.plot(x, y ,'ro')
+plt.axis([0,2.0,0, 15.0])
+plt.xlabel(r'$x$')
+plt.ylabel(r'$y$')
+plt.title(r'Random numbers ')
+plt.show()
+
+
+
+
+

Challenge: try to write a similar code for a Logistic Regression case.

+
+
+ + + + +
+ + + + +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/content.html b/doc/LectureNotes/_build/html/content.html new file mode 100644 index 000000000..df890a7ee --- /dev/null +++ b/doc/LectureNotes/_build/html/content.html @@ -0,0 +1,275 @@ + + + + + + + + Content in Jupyter Book — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ +
+
+ +
+ + + + + + + + + + + + + + +
+ + +
+ +
+
+
+
+
+ +
+ +
+

Content in Jupyter Book

+

There are many ways to write content in Jupyter Book. This short section +covers a few tips for how to do so.

+
+ + + + +
+ + +
+ + +
+ +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/genindex.html b/doc/LectureNotes/_build/html/genindex.html new file mode 100644 index 000000000..830fded6e --- /dev/null +++ b/doc/LectureNotes/_build/html/genindex.html @@ -0,0 +1,240 @@ + + + + + + + + Index — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ +
+
+ +
+ + + + + + + + + + + + +
+ + +
+ +
+
+
+
+
+ +
+ + +

Index

+ +
+ +
+ + +
+ + +
+ + +
+ +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/index.html b/doc/LectureNotes/_build/html/index.html new file mode 100644 index 000000000..de49afb2f --- /dev/null +++ b/doc/LectureNotes/_build/html/index.html @@ -0,0 +1,2 @@ + + diff --git a/doc/LectureNotes/_build/html/intro.html b/doc/LectureNotes/_build/html/intro.html new file mode 100644 index 000000000..6ade2635e --- /dev/null +++ b/doc/LectureNotes/_build/html/intro.html @@ -0,0 +1,484 @@ + + + + + + + + Applied Data Analysis and Machine Learning — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ +
+
+ +
+ + + + + + + + + + + + + + +
+ + + +
+
+
+
+ +
+ +
+

Applied Data Analysis and Machine Learning

+
+

Introduction

+

Probability theory and statistical methods play a central role in science. Nowadays we are +surrounded by huge amounts of data. For example, there are about one trillion web pages; more than one +hour of video is uploaded to YouTube every second, amounting to years of content every +day; the genomes of 1000s of people, each of which has a length of more than a billion base pairs, have +been sequenced by various labs and so on. This deluge of data calls for automated methods of data analysis, +which is exactly what machine learning aims at providing.

+
+
+

Learning outcomes

+

This course aims at giving you insights and knowledge about many of the central algorithms used in Data Analysis and Machine Learning. The course is project based and through various numerical projects, normally three, you will be exposed to fundamental research problems in these fields, with the aim to reproduce state of the art scientific results. Both supervised and unsupervised methods will be covered. The emphasis is on a frequentist approach, although we will try to link it with a Bayesian approach as well. You will learn to develop and structure large codes for studying different cases where Machine Learning is applied to, get acquainted with computing facilities and learn to handle large scientific projects. A good scientific and ethical conduct is emphasized throughout the course. More specifically, after this course you will

+
    +
  • Learn about basic data analysis, statistical analysis, Bayesian statistics, Monte Carlo sampling, data optimization and machine learning;

  • +
  • Be capable of extending the acquired knowledge to other systems and cases;

  • +
  • Have an understanding of central algorithms used in data analysis and machine learning;

  • +
  • Understand linear methods for regression and classification, from ordinary least squares, via Lasso and Ridge to Logistic regression;

  • +
  • Learn about neural networks and deep learning methods for supervised and unsupervised learning. Emphasis on feed forward neural networks, convolutional and recurrent neural networks;

  • +
  • Learn about about decision trees, random forests, bagging and boosting methods;

  • +
  • Learn about support vector machines and kernel transformations;

  • +
  • Reduction of data sets, from PCA to clustering;

  • +
  • Autoencoders and Reinforcement Learning;

  • +
  • Work on numerical projects to illustrate the theory. The projects play a central role and you are expected to know modern programming languages like Python or C++ and/or Fortran (Fortran2003 or later).

  • +
+
+
+

Prerequisites

+

Basic knowledge in programming and mathematics, with an emphasis on +linear algebra. Knowledge of Python or/and C++ as programming +languages is strongly recommended and experience with Jupiter notebook +is recommended. Required courses are the equivalents to the University +of Oslo mathematics courses MAT1100, MAT1110, MAT1120 and at least one +of the corresponding computing and programming courses INF1000/INF1110 +or MAT-INF1100/MAT-INF1100L/BIOS1100/KJM-INF1100. Most universities +offer nowadays a basic programming course (often compulsory) where +Python is the recurring programming language.

+
+
+

The course has two central parts

+
    +
  1. Statistical analysis and optimization of data

  2. +
  3. Machine learning

  4. +
+

These topics will be scattered thorughout the course and may not necessarily be taught separately. Rather, we will often take an approach (during the lectures and project/exercise sessions) where say elements from statistical data analysis are mixed with specific Machine Learning algorithms.

+
+

Statistical analysis and optimization of data

+

The following topics will be covered

+
    +
  • Basic concepts, expectation values, variance, covariance, correlation functions and errors;

  • +
  • Simpler models, binomial distribution, the Poisson distribution, simple and multivariate normal distributions;

  • +
  • Central elements of Bayesian statistics and modeling;

  • +
  • Gradient methods for data optimization,

  • +
  • Monte Carlo methods, Markov chains, Gibbs sampling and Metropolis-Hastings sampling;

  • +
  • Estimation of errors and resampling techniques such as the cross-validation, blocking, bootstrapping and jackknife methods;

  • +
  • Principal Component Analysis (PCA) and its mathematical foundation

  • +
+
+
+

Machine learning

+

The following topics will be covered:

+
    +
  • Linear Regression and Logistic Regression;

  • +
  • Neural networks and deep learning, including convolutional and recurrent neural networks

  • +
  • Decisions trees, Random Forests, Bagging and Boosting

  • +
  • Support vector machines

  • +
  • Bayesian linear and logistic regression

  • +
  • Boltzmann Machines

  • +
  • Unsupervised learning Dimensionality reduction, from PCA to cluster models

  • +
+

Hands-on demonstrations, exercises and projects aim at deepening your understanding of these topics.

+

Computational aspects play a central role and you are +expected to work on numerical examples and projects which illustrate +the theory and varous algorithms discussed during the lectures. We recommend strongly to form small project groups of 2-3 participants, if possible.

+
+
+
+

Required Technologies

+

Course participants are expected to have their own laptops/PCs. We use Git as version control software and the usage of providers like GitHub, GitLab or similar are strongly recommended.

+

We will make extensive use of Python as programming language and its +myriad of available libraries. You will find +Jupyter notebooks invaluable in your work. You can run R +codes in the Jupyter/IPython notebooks, with the immediate benefit of +visualizing your data. You can also use compiled languages like C++, +Rust, Julia, Fortran etc if you prefer. The focus in these lectures will be mainly +on Python.

+

If you have Python installed and you feel +pretty familiar with installing different packages, we recommend that +you install the following Python packages via pip as

+
    +
  • pip install numpy scipy matplotlib ipython scikit-learn mglearn sympy pandas pillow

  • +
+

For OSX users we recommend, after having installed Xcode, to +install brew. Brew allows for a seamless installation of additional +software via for example

+
    +
  • brew install python3

  • +
+

For Linux users, with its variety of distributions like for example the widely popular Ubuntu distribution, +you can use pip as well and simply install Python as

+
    +
  • sudo apt-get install python3

  • +
+
+

Python installers

+

If you don’t want to perform these operations separately and venture +into the hassle of exploring how to set up dependencies and paths, we +recommend two widely used distrubutions which set up all relevant +dependencies for Python, namely

+ +

which is an open source +distribution of the Python and R programming languages for large-scale +data processing, predictive analytics, and scientific computing, that +aims to simplify package management and deployment. Package versions +are managed by the package management system conda.

+ +

is a Python +distribution for scientific and analytic computing distribution and +analysis environment, available for free and under a commercial +license.

+

Furthermore, Google’s Colab:https://colab.research.google.com/notebooks/welcome.ipynb is a free Jupyter notebook environment that requires +no setup and runs entirely in the cloud. Try it out!

+
+
+

Useful Python libraries

+

Here we list several useful Python libraries we strongly recommend (if you use anaconda many of these are already there)

+
    +
  • NumPy:https://www.numpy.org/ is a highly popular library for large, multi-dimensional arrays and matrices, along with a large collection of high-level mathematical functions to operate on these arrays

  • +
  • The pandas:https://pandas.pydata.org/ library provides high-performance, easy-to-use data structures and data analysis tools

  • +
  • Xarray:http://xarray.pydata.org/en/stable/ is a Python package that makes working with labelled multi-dimensional arrays simple, efficient, and fun!

  • +
  • Scipy:https://www.scipy.org/ (pronounced “Sigh Pie”) is a Python-based ecosystem of open-source software for mathematics, science, and engineering.

  • +
  • Matplotlib:https://matplotlib.org/ is a Python 2D plotting library which produces publication quality figures in a variety of hardcopy formats and interactive environments across platforms.

  • +
  • Autograd:https://github.com/HIPS/autograd can automatically differentiate native Python and Numpy code. It can handle a large subset of Python’s features, including loops, ifs, recursion and closures, and it can even take derivatives of derivatives of derivatives

  • +
  • SymPy:https://www.sympy.org/en/index.html is a Python library for symbolic mathematics.

  • +
  • scikit-learn:https://scikit-learn.org/stable/ has simple and efficient tools for machine learning, data mining and data analysis

  • +
  • TensorFlow:https://www.tensorflow.org/ is a Python library for fast numerical computing created and released by Google

  • +
  • Keras:https://keras.io/ is a high-level neural networks API, written in Python and capable of running on top of TensorFlow, CNTK, or Theano

  • +
  • And many more such as pytorch:https://pytorch.org/, Theano:https://pypi.org/project/Theano/ etc

  • +
+
+
+
+
+
+
+
+ + + + +
+ + + + +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/objects.inv b/doc/LectureNotes/_build/html/objects.inv new file mode 100644 index 000000000..93ff2cbc4 --- /dev/null +++ b/doc/LectureNotes/_build/html/objects.inv @@ -0,0 +1,7 @@ +# Sphinx inventory version 2 +# Project: Python +# Version: +# The remainder of this file is compressed using zlib. +xڅMn! h*g]U"Ei/( DܾD;x0`[GdMl'H#f7G Ȣfsg[:OtR31 (m@ +^1"l=-ۣ4iɰK{NZ4A lTȎ.v{ZeL}ӘRd@3C `ڗE'8B YF_~@M"q=vҪCǰհHxJ?+*K \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/reports/chapter1.log b/doc/LectureNotes/_build/html/reports/chapter1.log new file mode 100644 index 000000000..5d9252bb7 --- /dev/null +++ b/doc/LectureNotes/_build/html/reports/chapter1.log @@ -0,0 +1,31 @@ +Traceback (most recent call last): + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/jupyter_cache/executors/utils.py", line 51, in single_nb_execution + executenb( + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 1087, in execute + return NotebookClient(nb=nb, resources=resources, km=km, **kwargs).execute() + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 74, in wrapped + return just_run(coro(*args, **kwargs)) + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 53, in just_run + return loop.run_until_complete(coro) + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/asyncio/base_events.py", line 616, in run_until_complete + return future.result() + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 540, in async_execute + await self.async_execute_cell( + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 832, in async_execute_cell + self._check_raise_for_error(cell, exec_reply) + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 740, in _check_raise_for_error + raise CellExecutionError.from_cell_and_msg(cell, exec_reply['content']) +nbclient.exceptions.CellExecutionError: An error occurred while executing the following cell: +------------------ +import numpy as np +x = np.log(np.array([4.0, 7.0, 8.0]) +print(x) +------------------ + + File "", line 3 + print(x) + ^ +SyntaxError: invalid syntax + +SyntaxError: invalid syntax (, line 3) + diff --git a/doc/LectureNotes/_build/html/reports/chapter2.log b/doc/LectureNotes/_build/html/reports/chapter2.log new file mode 100644 index 000000000..825b97e1f --- /dev/null +++ b/doc/LectureNotes/_build/html/reports/chapter2.log @@ -0,0 +1,104 @@ +Traceback (most recent call last): + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/jupyter_cache/executors/utils.py", line 51, in single_nb_execution + executenb( + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 1087, in execute + return NotebookClient(nb=nb, resources=resources, km=km, **kwargs).execute() + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 74, in wrapped + return just_run(coro(*args, **kwargs)) + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 53, in just_run + return loop.run_until_complete(coro) + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/asyncio/base_events.py", line 616, in run_until_complete + return future.result() + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 540, in async_execute + await self.async_execute_cell( + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 832, in async_execute_cell + self._check_raise_for_error(cell, exec_reply) + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 740, in _check_raise_for_error + raise CellExecutionError.from_cell_and_msg(cell, exec_reply['content']) +nbclient.exceptions.CellExecutionError: An error occurred while executing the following cell: +------------------ +%matplotlib inline + +from numpy import * +from numpy.random import randint, randn +from time import time +import matplotlib.mlab as mlab +import matplotlib.pyplot as plt + +# Returns mean of bootstrap samples +def stat(data): + return mean(data) + +# Bootstrap algorithm +def bootstrap(data, statistic, R): + t = zeros(R); n = len(data); inds = arange(n); t0 = time() + # non-parametric bootstrap + for i in range(R): + t[i] = statistic(data[randint(0,n,n)]) + + # analysis + print("Runtime: %g sec" % (time()-t0)); print("Bootstrap Statistics :") + print("original bias std. error") + print("%8g %8g %14g %15g" % (statistic(data), std(data),mean(t),std(t))) + return t + + +mu, sigma = 100, 15 +datapoints = 10000 +x = mu + sigma*random.randn(datapoints) +# bootstrap returns the data sample +t = bootstrap(x, stat, datapoints) +# the histogram of the bootstrapped data +n, binsboot, patches = plt.hist(t, 50, normed=1, facecolor='red', alpha=0.75) + +# add a 'best fit' line +y = mlab.normpdf( binsboot, mean(t), std(t)) +lt = plt.plot(binsboot, y, 'r--', linewidth=1) +plt.xlabel('Smarts') +plt.ylabel('Probability') +plt.axis([99.5, 100.6, 0, 3.0]) +plt.grid(True) + +plt.show() +------------------ + +--------------------------------------------------------------------------- +AttributeError Traceback (most recent call last) + in  + 31 t = bootstrap(x, stat, datapoints) + 32 # the histogram of the bootstrapped data +---> 33 n, binsboot, patches = plt.hist(t, 50, normed=1, facecolor='red', alpha=0.75) + 34  + 35 # add a 'best fit' line + +~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/pyplot.py in hist(x, bins, range, density, weights, cumulative, bottom, histtype, align, orientation, rwidth, log, color, label, stacked, data, **kwargs) + 2683 orientation='vertical', rwidth=None, log=False, color=None, + 2684 label=None, stacked=False, *, data=None, **kwargs): +-> 2685 return gca().hist( + 2686 x, bins=bins, range=range, density=density, weights=weights, + 2687 cumulative=cumulative, bottom=bottom, histtype=histtype, + +~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/__init__.py in inner(ax, data, *args, **kwargs) + 1445 def inner(ax, *args, data=None, **kwargs): + 1446 if data is None: +-> 1447 return func(ax, *map(sanitize_sequence, args), **kwargs) + 1448  + 1449 bound = new_sig.bind(ax, *args, **kwargs) + +~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/axes/_axes.py in hist(self, x, bins, range, density, weights, cumulative, bottom, histtype, align, orientation, rwidth, log, color, label, stacked, **kwargs) + 6813 if patch: + 6814 p = patch[0] +-> 6815 p.update(kwargs) + 6816 if lbl is not None: + 6817 p.set_label(lbl) + +~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/artist.py in update(self, props) + 994 func = getattr(self, f"set_{k}", None) + 995 if not callable(func): +--> 996 raise AttributeError(f"{type(self).__name__!r} object " + 997 f"has no property {k!r}") + 998 ret.append(func(v)) + +AttributeError: 'Rectangle' object has no property 'normed' +AttributeError: 'Rectangle' object has no property 'normed' + diff --git a/doc/LectureNotes/_build/html/reports/chapter4.log b/doc/LectureNotes/_build/html/reports/chapter4.log new file mode 100644 index 000000000..f2d559b52 --- /dev/null +++ b/doc/LectureNotes/_build/html/reports/chapter4.log @@ -0,0 +1,89 @@ +Traceback (most recent call last): + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/jupyter_cache/executors/utils.py", line 51, in single_nb_execution + executenb( + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 1087, in execute + return NotebookClient(nb=nb, resources=resources, km=km, **kwargs).execute() + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 74, in wrapped + return just_run(coro(*args, **kwargs)) + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/util.py", line 53, in just_run + return loop.run_until_complete(coro) + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/asyncio/base_events.py", line 616, in run_until_complete + return future.result() + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 540, in async_execute + await self.async_execute_cell( + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 832, in async_execute_cell + self._check_raise_for_error(cell, exec_reply) + File "/Users/mhjensen/opt/anaconda3/lib/python3.8/site-packages/nbclient/client.py", line 740, in _check_raise_for_error + raise CellExecutionError.from_cell_and_msg(cell, exec_reply['content']) +nbclient.exceptions.CellExecutionError: An error occurred while executing the following cell: +------------------ +%matplotlib inline + +# Common imports +import os +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +from sklearn.linear_model import LinearRegression, Ridge, Lasso +from sklearn.model_selection import train_test_split +from sklearn.utils import resample +from sklearn.metrics import mean_squared_error +from IPython.display import display +from pylab import plt, mpl +plt.style.use('seaborn') +mpl.rcParams['font.family'] = 'serif' + +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("chddata.csv"),'r') + +# Read the chd data as csv file and organize the data into arrays with age group, age, and chd +chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD')) +chd.columns = ['ID', 'Age', 'Agegroup', 'CHD'] +output = chd['CHD'] +age = chd['Age'] +agegroup = chd['Agegroup'] +numberID = chd['ID'] +display(chd) + +plt.scatter(age, output, marker='o') +plt.axis([18,70.0,-0.1, 1.2]) +plt.xlabel(r'Age') +plt.ylabel(r'CHD') +plt.title(r'Age distribution and Coronary heart disease') +plt.show() +------------------ + +--------------------------------------------------------------------------- +FileNotFoundError Traceback (most recent call last) + in  + 38 plt.savefig(image_path(fig_id) + ".png", format='png') + 39  +---> 40 infile = open(data_path("chddata.csv"),'r') + 41  + 42 # Read the chd data as csv file and organize the data into arrays with age group, age, and chd + +FileNotFoundError: [Errno 2] No such file or directory: 'DataFiles/chddata.csv' +FileNotFoundError: [Errno 2] No such file or directory: 'DataFiles/chddata.csv' + diff --git a/doc/LectureNotes/_build/html/schedule.html b/doc/LectureNotes/_build/html/schedule.html new file mode 100644 index 000000000..fda9aaf5b --- /dev/null +++ b/doc/LectureNotes/_build/html/schedule.html @@ -0,0 +1,288 @@ + + + + + + + + Teaching schedule with links to material — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ +
+
+ +
+ + + + + + + + + + + + + + +
+ + +
+ +
+
+
+
+
+ +
+ + + + + + +
+ + + + +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/search.html b/doc/LectureNotes/_build/html/search.html new file mode 100644 index 000000000..2e8f6eee3 --- /dev/null +++ b/doc/LectureNotes/_build/html/search.html @@ -0,0 +1,259 @@ + + + + + + + + Search — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ +
+
+ +
+ + + + + + + + + + + + +
+ + +
+ +
+
+
+
+
+ +
+ +

Search

+
+ +

+ Please activate JavaScript to enable the search + functionality. +

+
+

+ Searching for multiple words only shows matches that contain + all words. +

+
+ + + +
+ +
+ +
+ +
+ + +
+ + +
+ +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/searchindex.js b/doc/LectureNotes/_build/html/searchindex.js new file mode 100644 index 000000000..057cd160b --- /dev/null +++ b/doc/LectureNotes/_build/html/searchindex.js @@ -0,0 +1 @@ +Search.setIndex({docnames:["chapter1","chapter2","chapter3","chapter4","content","intro","schedule","teachers","textbooks"],envversion:{"sphinx.domains.c":2,"sphinx.domains.changeset":1,"sphinx.domains.citation":1,"sphinx.domains.cpp":3,"sphinx.domains.index":1,"sphinx.domains.javascript":2,"sphinx.domains.math":2,"sphinx.domains.python":2,"sphinx.domains.rst":2,"sphinx.domains.std":1,"sphinx.ext.intersphinx":1,sphinx:56},filenames:["chapter1.ipynb","chapter2.ipynb","chapter3.ipynb","chapter4.ipynb","content.md","intro.md","schedule.md","teachers.md","textbooks.md"],objects:{},objnames:{},objtypes:{},terms:{"000000":2,"00000000e":2,"001":3,"00727646693":0,"0086649156":0,"01043351":0,"0123897":2,"012390":2,"03435975":2,"034360":2,"03494744740413562":2,"057678":2,"059991":2,"061016":2,"062435":2,"062978":2,"063987":2,"064814":2,"064967":2,"065032":2,"065108":2,"065469":2,"065995":2,"066970":2,"067002":2,"067071":2,"067277":2,"067321":2,"067805":2,"0684659365902349":2,"068481":2,"068545":2,"069095":2,"069384":2,"069504":2,"069637":2,"069640":2,"069793":2,"069824":2,"069875":2,"070113":2,"071062":2,"07106781e":2,"0713":0,"071758":2,"071774":2,"071804":2,"072069":2,"072268":2,"072399":2,"072514":2,"072554":2,"072629":2,"072824":2,"072999":2,"073058":2,"073451":2,"073564":2,"073761":2,"074261":2,"074593":2,"074724":2,"075069":2,"075149":2,"075299":2,"075348":2,"075401":2,"075503":2,"075775":2,"075932":2,"076146":2,"076323":2,"076602":2,"076979":2,"077022":2,"077152":2,"077490":2,"077804":2,"077858":2,"078039":2,"078366":2,"078437":2,"078639":2,"078769":2,"079293":2,"07944154":0,"079482":2,"079726":2,"080521":2,"080887":2,"081135":2,"081453":2,"081476":2,"081717":2,"081951":2,"081952":2,"08248290e":2,"082574":2,"082602":2,"082702":2,"082762":2,"082882":2,"083219":2,"083307":2,"084869":2,"085171":2,"085647":2,"085972":2,"087047":2,"087104":2,"087489":2,"088312":2,"088620":2,"088986":2,"090255":2,"091332":2,"093536":2,"094412":2,"094670":2,"095035":2,"100":[0,1,2,3,7],"1000":[0,3,5],"10000":1,"101207":2,"103129":2,"10312915":2,"10484789":2,"104848":2,"10659197":0,"10898112e":2,"10x":0,"111":3,"124":0,"133":3,"136239":1,"1445":1,"1446":1,"1447":1,"1448":1,"1449":1,"14g":1,"150205":1,"150467":1,"15493691":2,"15g":1,"16328353":0,"16496581e":2,"18404906e":2,"1940":0,"1970":0,"1979":1,"1cm":0,"200":0,"2004":3,"2016":0,"2018":1,"2020":7,"207":1,"209":1,"227172":2,"22717222":2,"250":3,"25000":0,"26202974":2,"262030":2,"2683":1,"2684":1,"2685":1,"2686":1,"2687":1,"26881845":0,"287181":2,"28718135":2,"2890":0,"2931":0,"2968":0,"2980":0,"2990":0,"30000":0,"30075596":0,"3155":1,"33066907e":2,"333":3,"3436":0,"3437":0,"3588324":0,"35932784":2,"359328":2,"36630178":2,"366302":2,"38629436":0,"3864659735909566":2,"39431833":2,"4000":8,"462":3,"48257387":7,"4877836":2,"487784":2,"4940954":0,"500":[1,3],"506":0,"50j":3,"55248701":0,"56083552":2,"56326636":0,"56536":0,"593139":2,"59313927":2,"625":3,"626603":2,"62660334":2,"62761155":2,"627612":2,"65939208e":2,"676675":2,"67667546":2,"6813":1,"6814":1,"6815":1,"6816":1,"6817":1,"72320998":2,"723210":2,"74988099":2,"749881":2,"765":3,"772b904ae9cb":1,"77350269e":2,"77593":1,"800":3,"803981":2,"8039812":2,"896222705525891":2,"90178573":2,"901786":2,"9054":1,"90745348":0,"9154":1,"924408":2,"92440839":2,"931":0,"939":0,"94591015":0,"95648659":2,"9646":1,"97644907":0,"9780387310732":8,"9780387848570":8,"9781492032632":8,"989696":2,"994":1,"995":1,"996":1,"99624087":2,"996241":2,"997":1,"998":1,"break":0,"byte":0,"case":[0,1,2,5],"catch":0,"class":[0,1,3],"default":[0,3],"f\u00f8470":7,"final":[0,1,2,6,7],"float":0,"function":[1,5],"import":[1,2,3],"int":[0,2,3],"long":[0,3],"new":[0,1,2,3],"public":[0,5],"return":[0,1,2,3],"short":4,"super":2,"switch":0,"true":[0,1,3],"try":[0,3,5],"var":[1,2],"while":[0,1,2,3],AGE:0,Age:3,And:[0,1,5],Being:3,But:[0,1],CAS:0,DIS:0,Doing:3,EoS:[0,1],FYS:6,For:[0,1,2,3,5,8],Going:2,Ising:2,Not:[0,1,2],OLS:[0,1,2],One:[1,2,3],PCs:5,Such:1,That:[0,3],The:[6,7,8],Then:[0,1,3],There:[0,2,4,6,7],These:[0,2,5],Useful:1,Using:[0,1,2],With:[0,1,2],__doc__:1,__init__:1,__name__:1,_auto1:[2,3],_ax:1,_datafram:0,_lambda:0,a77d5ac269b2:3,a_0:0,a_1a:0,a_2a:0,a_3:0,a_3a:0,a_4:0,a_4a:0,a_i:0,abil:0,abl:[0,2,3],about:[0,1,2,3,5,8],abov:[0,1,2,3],abovement:1,abs:0,abscissa:3,absolut:[0,1,2],acccess:0,accept:0,access:[0,8],accid:1,accompani:0,accord:[0,1,3],account:0,accur:1,accuraci:[0,2,3],accuracy_scor:0,achiev:[0,1],acquaint:5,acquir:5,acr:0,across:[0,1,5],activ:[0,6],actual:[0,1,2],adapt:[1,3,8],add:[0,1,2],add_subplot:3,added:[0,2,3],adding:1,addit:[0,1,3,5,7,8],addition:3,address:[3,8],adjust:3,admir:0,advanc:[1,8],advantag:[1,3],afecionado:0,affect:2,affin:0,aficionado:0,african:0,after:[0,1,2,3,5],afterward:0,again:[0,1,3],against:3,age:[0,3],agegroup:3,agegroupmean:3,aim:[0,1,3,5],algebra:[0,2,3,5],algorithm:[0,1,3,5,8],align:[0,1,2,3],all:[0,1,2,3,5,6,7],allevi:3,allow:[0,1,3,5,7],almost:[0,1,3],along:[0,1,2,5],alpha:[0,1,3],alpha_i:3,alpha_k:3,alpha_opt:3,alreadi:[0,5],also:[0,1,2,3,5,6,7,8],altern:0,although:[1,5],alwai:[0,1,2,3],ame2016:0,american:0,among:0,amount:[1,5],anaconda3:1,anaconda:[0,5],analys:[1,2],analysi:[0,1,2,3,8],analyt:[0,1,2,3,5],analyz:[0,2],angl:0,ani:[0,1,2],annot:[0,3],anoth:[0,1,3],ansatz:0,answer:[0,1],anytim:7,apart:3,api:[0,5],appear:0,append:[0,1,3],appli:[0,1,3,8],applic:[0,1,3,8],approach:[0,1,2,5,8],appropri:1,approx:[0,3],approxim:[0,1,2,3],apt:[0,5],aragorn:0,arang:[0,1,3],arbitrari:3,arbitrarili:0,architectur:8,area:[0,8],arg:1,argument:0,aris:[0,1,3],arithmet:0,around:[0,1],arrai:[1,2,3,5],arriv:0,art:5,articl:[0,1,2],artifici:[0,3,8],artist:1,asarrai:0,ascii:0,ask:1,aspect:[0,5],assembl:0,assess:1,assign:[0,3,6],associ:[0,1],assum:[0,1,2,3],assumpt:[0,1],ast:[0,1],asymmetri:0,asymptot:1,atom:0,attempt:[0,3],attend:6,attent:0,attract:0,attribut:0,attributeerror:1,audi:0,aurelien:8,author:0,authour:0,autoencod:5,autograd:[0,5],autom:5,automag:0,automat:[0,5],autonom:8,avail:[0,1,5,6],averag:[0,1,7],avoid:[0,1,2],award:7,axes3d:3,axes:[1,3],axi:[0,1,3],axlabel:0,b_1:3,b_5:3,b_i:0,b_ia_:0,b_k:3,bachelor:6,back:[0,2],background:8,bag:5,baggin:0,balanc:1,band:0,bandwidth:0,bar:0,barber:8,base:[0,2,3,5,7,8],basi:[0,2,3],basic:[2,5],batch:3,bay:3,bayesian:[5,8],becaus:[0,1,3],becom:[0,1,2,3],been:[0,1,5],befor:[0,1,2,3],beforehand:0,begin:[0,1,2,3],behav:[1,3],behavior:[0,3],behind:[0,3],being:[0,2,3],belong:3,below:[0,1,2,3,7],benefit:[0,3,5],benign:3,best:[0,1,3,7],beta:[0,1,2,3],beta_0:[0,3],beta_0x_:0,beta_1:[0,3],beta_1x_0:0,beta_1x_1:[0,3],beta_1x_2:0,beta_1x_:0,beta_1x_i:3,beta_2:0,beta_2x_0:0,beta_2x_1:0,beta_2x_2:[0,3],beta_2x_:0,beta_:[0,3],beta_i:[0,2],beta_j:[0,3],beta_k:3,beta_linreg:3,beta_p:3,beta_px_p:3,better:0,between:[0,1,2,3],beyond:[0,3],bia:[0,2],bias:1,big:[0,1],bilbo:0,billion:5,bin:[0,1,3],binari:[0,3],bind:[0,1],binomi:5,binsboot:1,bioinformat:0,biolog:8,bios1100:5,bird:0,birth:0,bishop:8,bit:0,bla:0,block:[0,1,5],blue:0,bmatrix:[0,2,3],bodi:0,boldfac:0,boldsymbol:[0,1,2,3],boltzmann:5,book:8,boost:5,bootstrap:5,borrow:0,boston_dataset:0,both:[0,1,2,3,5],bottl:3,bottom:1,bound:[0,1],boyd:3,brain:3,breast:3,brew:[0,5],briefli:0,bring:0,broad:0,browser:0,brute:2,build:[0,1],built:[0,1],busi:0,c_i:3,cal:3,calcul:[0,1,2,3],call:[0,1,2,3,5,8],callabl:1,cambridg:[3,8],can:[0,1,2,3,5,6,8],cancel:0,cancerpd:3,cannot:[0,2,3,6],canopi:[0,5],capabl:[0,5],capita:0,card:[0,3],carefulli:3,carlo:[0,1,5,8],carri:[1,3],casella:8,categor:0,categori:[0,3],caus:[1,2],causal:0,cdot:[0,1,3],celebr:3,center:[0,1,3],centr:8,central:[0,1],certain:[0,1,3],cha:0,chain:5,challeng:3,chanc:3,chang:[0,1,2,3],chapter:[1,8],charact:[0,2],characterist:0,charg:0,charl:0,chd:3,chddata:3,cheap:2,cheaper:3,check:[0,3],choic:[0,1,3],choleski:2,choos:[1,3],chosen:[0,1,3],christian:8,christoph:8,circl:0,circumv:2,classic:[0,3],classif:[0,1,3,5,8],classifi:3,clearli:[1,3],clf3:0,clf:0,clf_ridg:0,close:[0,1,3],closest:3,closur:[0,5],cloud:[0,5],cluster:[0,1,5],cmap:0,cntk:[0,5],code:[1,2,5,8],coef:0,coef_:[0,3],coeffici:[0,1,3],coerc:[0,1],col:0,colab:[0,5],colinear:0,collect:[0,1,5,8],collinear:2,color:[0,1],column:[0,1,2,3],com:[5,8],combin:[1,3],come:[0,2,3],comma:0,command:0,commerci:[0,5],commod:0,common:[0,1,2,3],commonli:[1,3],commun:0,compact:[0,1,2,3],compar:[0,1,2,3],compat:3,compet:0,compil:[0,5],complet:0,complex:1,complic:[0,1,3],compon:[0,1,2,3,5],compphys:6,compress:0,compris:1,compromis:2,compulsori:5,comput:[0,1,2,5,6,7,8],computation:[1,3],concav:3,concentr:0,concept:[0,5],conceptu:3,concern:3,concic:0,conclud:0,conda:[0,5],condit:[0,1,2],conduct:5,confid:[0,1,3],confus:1,conjugaci:3,connect:[0,3],consequ:[1,2,3],conserv:2,consid:[0,1,2,3],consider:[0,3],consist:[0,1,3],constant:[0,3],constitu:0,constitut:1,constrain:3,constraint:[2,3],construct:[0,1,2,3,8],contact:0,contain:[0,1,3,8],contemporari:8,content:[0,5],context:[1,3],continu:[0,1,3],contour:3,contribut:[0,2],contributor:0,control:[0,5],conveni:[0,1,3],converg:[2,3],convert:[0,2],convinc:3,convolut:5,coordin:2,coorel:0,coronari:3,corr:[0,2,3],correalt:2,correct:[0,2],correctli:1,correl:[0,3,5],correlation_matrix:[0,2,3],correspond:[0,1,5],cos:[0,1],cosin:1,cost:[0,1,2],could:[0,1,2,3],coulomb:0,count:[0,6,7],countor:3,cours:[0,6],cov:[0,1,2],cov_xi:2,cov_xx:2,cov_yi:2,covari:[0,3,5],covariance_matrix:2,cover:[0,4,5,8],covert:0,covid:7,creat:[0,5],create_x:[0,2],credit:[0,3],crim:0,crime:0,criteria:0,criterion:3,cross:[0,3,5],cross_val_scor:1,cross_valid:3,crossvalid:1,csr_matrix:0,csv:[0,1,3],cubic:0,cumsum:0,cumul:1,current:3,curs:0,curv:3,curvatur:3,cvxbook:3,cyber:8,d_f:3,dagger:[0,2],dai:5,dat:0,dat_id:[0,1,3],data:[1,2,8],data_id:[0,1,3],data_panda:0,data_path:[0,1,3],databas:0,datafil:[0,1,3],datafram:[0,2,3],datapoint:[0,1,2,3],dataset:[0,1,3],date:0,david:8,deal:[0,3],debt:3,debug:1,decad:0,decai:[0,3],decid:1,decim:0,decis:[0,5,8],decisiontreeregressor:0,declar:0,decompos:2,decomposit:0,decompost:2,decreas:[1,3],deduc:0,deep:[3,5,8],deepen:5,def:[0,1,2,3],defect:2,defici:2,defin:[0,1,2,3],definit:[1,2,3],degre:[1,2],delet:1,deliv:[0,6],delta:0,delta_:0,delta_h:0,delta_n:0,delug:5,delv:0,demand:3,demonstr:[0,1,2,3,5],denot:[1,3],densiti:[0,1],depart:7,depend:[0,1,2,3,5],deploy:[0,5],deriv:[0,1,2,5],descend:2,descent:0,describ:[0,1],descript:0,design:[0,1,2,3],designmatrix:0,desir:[0,2,3],det:2,detail:[0,3],determin:[0,3],determinist:3,develop:[0,5],deviat:[0,1],df1:0,diag:2,diagon:[0,2,3],diagonaliz:2,dictionari:0,did:[0,2,3],differ:[0,1,2,5],differenti:[0,3,5],difficult:1,difficulti:3,digit:[0,6],dimens:[0,2],dimension:[0,1,2,3,5],dimensionless:0,direct:[0,3],directori:3,disadvantag:0,discard:1,disciplin:0,discourag:3,discret:3,discrimin:3,discuss:[0,1,2,3,5],diseas:3,disk:0,disord:3,displai:[0,1,3],displaystyl:[0,2],disregard:0,distanc:[0,6],distinct:3,distinguish:[0,3],distplot:0,distribut:[0,1,3,5],distrubut:[0,5],dive:[0,2],diverg:3,divid:[0,1],divis:1,dna:3,dnn:0,dnn_scikit:0,doc:[5,6],doconc:0,doe:[0,1,2,3],doing:[0,1],domain:3,domin:0,don:[0,5],done:[0,1,2],dot:[0,2,3],doubl:0,down:[0,3],download:[0,8],draw:[1,3],drawback:[0,3],drawn:[0,1,3],drop:[0,1,2],dropna:[0,1],dtype:0,dub:0,due:[1,2,3,6],dummi:0,dure:[0,5],dwell:0,each:[0,1,2,3,5,6,7],eapprox:0,earlier:[0,3],easi:[0,1,2,3,5],easier:1,easiest:3,easili:[0,2,3],eastern:7,ebind:0,econometr:0,ecosystem:[0,5],ect:6,edgecolor:1,edit:0,edu:3,educ:0,effici:[0,3,5],efron:1,eig:[0,2,3],eigenpair:2,eigenvalu:[0,2,3],eigenvector:[2,3],eight:0,eigval:0,eigvalu:3,eigvec:0,eigvector:3,eispack:0,either:[0,1,2,3],electr:0,element:[1,2,3,5,8],elementari:0,elessar:0,els:3,email:[6,7],embed:0,embodi:1,emner:8,emphas:[0,5],emphasi:[0,5,8],emploi:[0,1,3],employ:0,empti:1,encompass:0,encount:[0,2,3],end:[0,1,2,3],energi:[0,1],eng:8,engin:[0,5],english:8,enough:[1,3],ensur:[0,1,3],enter:2,enthought:[0,5],entir:[0,3,5],entiti:0,entri:[0,2],entropi:[0,3],enumer:0,environ:[0,5,8],eol:0,eosfit:0,epoch:[0,3],epsilon:[0,1,3],epsilon_0:0,epsilon_1:0,epsilon_2:0,epsilon_:0,epsilon_i:0,eqnarrai:1,equal:[0,1,2,3],equat:[1,2],equiv:3,equival:[0,2,5],eriador:0,eridg:0,err:0,err_:1,errat:3,errno:3,error:[0,1,2,3,5],escap:3,essenti:[0,2],estim:[0,1,2,3,5],estimated_mse_fold:1,estimated_mse_kfold:1,estimated_mse_sklearn:1,eta0:3,eta:[0,3],eta_v:0,etc:[2,3,5],ethic:5,etsim:1,euclidean:0,evalu:[0,1,2,3],even:[0,1,3,5],event:3,eventu:[1,2,3,7],everi:[0,1,3,5],everywher:3,evolv:0,exact:[0,2,3],exactli:[0,5],examin:1,exampl:[1,2,5,8],excel:[0,8],except:1,excess:0,excit:0,exclud:1,exclus:[0,1],execut:2,exemplifi:3,exercis:[0,5,6],exhaust:1,exhibit:0,exist:[0,1,3,8],exit:2,exp:[0,1,2,3],expand:[2,3],expans:[0,2,3],expect:[0,1,2,3,5],expens:[1,3],experi:[0,1,3,5],experiment:[0,1],explain:[0,3],explanatori:0,explicit:0,explicitli:0,exploit:0,explor:[0,3,5],exponenti:[0,3],expos:5,express:[0,1,2],extend:5,extens:[0,5],extent:[0,1],extra:2,extract:[0,2,3],extrapol:0,extrem:[0,3],extremum:3,eye:[0,3],f11:0,f12:0,f13:0,f1d:3,f6d7a289d493:0,f_1:3,f_2:3,f_i:1,face:3,facecolor:1,facil:5,fact:[0,2,3],factor:[0,2],fail:[1,3,7],failur:3,fall:[6,7],fals:[0,1,3],famili:[0,3],familiar:[0,5],famou:1,far:[0,2,3],fast:[0,1,3,5],fastest:3,favor:3,featur:[1,2,3,5],feature_nam:[0,3],feed:[0,5],feel:[0,5,7],feet:0,few:4,fewer:0,field:[0,5],fifth:0,fig:[0,3],fig_id:[0,1,3],figsiz:[0,1,3],figur:[0,1,3,5],figure_id:[0,1,3],figurefil:[0,1,3],file:[0,1,3],filenam:0,filenotfounderror:3,fill:2,financ:0,find:[0,1,2,3,5],finit:[1,2],first:[0,1,2,8],fit:[1,2,3],fit_intercept:1,fit_transform:[0,1],fiti:0,five:0,fix:[0,1,3],flat:3,flexibl:[0,1],float64:0,flop:2,fmesh:3,focu:[0,1,5,8],focus:3,fold:1,folder:0,follow:[0,1,2,3,5,7],font:[0,3],forc:[0,2],forest:[0,5],form:[0,1,2,3,5],format:[0,1,3,5],formatstrformatt:3,formula:3,fortran2003:5,fortran:[0,5],fortun:0,forward:[0,1,5],found:[1,3],foundat:5,four:6,fourier:0,fourth:0,frac:[0,1,2,3],frame:3,frank:2,frankefunct:[0,2],free:[0,5,7,8],freedom:2,freeli:0,frequenc:[1,3],frequent:[0,3],frequentist:5,friedman:8,frodo:0,from:[0,1,2,3,5,7],front:0,fs20:7,fulfil:2,full:[0,2,3],fulli:[1,6],fun:[0,5],func:1,fundament:[1,5],further:0,furthermor:[0,1,2,3,5],futur:0,fys:7,gain:2,galleri:0,gamge:0,gamma:[0,3],gamma_:0,gamma_i:0,gamma_j:3,gamma_k:3,gamma_x:0,gaussian:1,gave:3,gca:[1,3],gender:0,gener:[0,1,2,3,8],genom:5,geometr:0,georg:8,geq:[2,3],geron:8,get:[0,1,3,5],getattr:1,gibb:5,git:5,github:[0,5,6],gitlab:5,give:[0,2,3,5,8],given:[0,1,2,3],global:3,goal:[0,3],goe:[0,1,2,3],going:[0,1,2,3],golden:3,gone:2,good:[0,3,5,8],googl:[0,5],grade:6,gradient:[0,5],graph:3,graphic:0,grasp:0,great:3,greater:3,green:0,grid:[1,3],grossli:3,ground:0,group:[0,1,3,5,6,7],groupbi:0,growth:0,guarante:[0,3],guess:3,h_1:3,h_2:3,had:[0,1,2,3],hand:[0,3,5,8],handl:5,handwritten:3,happen:[2,3],hard:3,hardcopi:[0,5],harder:0,has:[0,1,2,3],hassl:[0,5],hast:5,hasti:8,hat:[0,2,3],have:[0,1,2,3,5],hdf5:0,head:0,header:0,hear:0,heart:[0,3],heatmap:[0,3],heavili:0,henc:[0,1,2,3],her:3,here:[0,1,2,3,5,8],hereaft:0,hermitian:0,hessenberg:0,hidden_layer_s:0,high:[0,1,2,3,5],higher:[0,1,3],highli:[0,2,5],highwai:0,hint:3,hip:5,hire:0,his:3,hist:[1,3],histogram:[0,1,3],histor:3,histtyp:1,hjorth:7,hoc:2,hoff:8,hold:[1,3],holder:0,home:[0,8],hopefulli:0,hors:3,hour:[5,6,7],how:[0,1,2,3,4,5,8],howev:[0,1,2,3],hspace:0,html:[0,5,6,8],http:[0,3,5,6,8],huang:0,huber:0,huge:5,human:0,hybrid:6,hydrogen:0,hyperparamet:[0,2],i_1:1,i_2:1,idea:[0,1,3],ideal:[0,1],idem:1,ident:[1,2],identifi:[0,3],ifi:8,ifs:[0,5],ignor:0,iii:0,illustr:[3,5],imag:8,image_path:[0,1,3],immedi:[0,5],implement:[0,3],impli:[0,1,2,3],imposs:2,impress:0,improv:2,in3050:8,in4080:8,in4300:8,in5400:8,inaccur:3,includ:[0,3,5,7],include_bia:1,increas:[0,1],ind:1,inde:[0,2],independ:[0,1,2,3],index:[0,5,8],index_col:0,indic:[0,2],indispens:1,individu:[3,7],indu:0,inequaltii:3,inf1000:5,inf1100:5,inf1100l:5,inf1110:5,inf3000:8,inf4490:8,inf5860:8,infer:[0,1,8],infil:[0,1,3],infin:[1,2,3],influenc:1,info:0,inform:[0,1,3,8],infti:3,ingeni:3,ingredi:0,inher:1,inherit:0,initi:[0,1,3],inlin:[0,1,3],inner:[1,3],innov:8,input:[0,1,2,3],insid:[0,3],insight:[0,2,5,8],insist:3,inspir:[0,8],instanc:[0,1,3],instead:[0,1,2,3],instruct:7,integ:0,integr:0,intellig:[0,8],interact:[0,5],intercept:[0,3],intercept_:[0,3],interest:[0,1,2,3],interfac:0,interior:0,interpr:2,interpret:0,interv:[0,1,3],intial:3,intimid:0,intract:0,intrins:0,introduc:[0,3],introduct:[3,8],introductori:[0,8],intuit:[0,1],inv:[0,2,3],invalid:0,invalu:[0,3,5],invd:2,invers:[0,2,3],invert:[0,2,3],invok:0,involv:[0,1,3],ipynb:[0,5],ipython:[0,1,3,5],irreduc:1,irrelev:2,irrespect:0,isnul:0,it_arrai:3,item:0,items:0,iter:1,its:[0,1,2,3,5,8],itself:1,jackknif:[1,5],jacobian:3,jensen:7,jerom:8,join:[0,1,3],judg:3,julia:5,jupit:5,jupyt:[0,5],just:[0,1,2,3],justif:0,keep:[0,2,3],keepdim:1,kei:[0,8],kera:[0,5],kernel:[0,5],kev:0,kevin:8,keyword:0,kfold:1,kick:3,kind:0,kjm:5,know:[0,2,3,5],knowledg:[0,5],known:[1,3,8],kondev:0,kwarg:1,kwown:0,l_1:3,l_2:3,lab:5,label:[0,1,2,3,5],labelencod:3,laboratori:6,lack:0,lambda:[0,1,2,3],lambda_1:2,lambda_n:2,land:0,landscap:3,langl:0,languag:[0,5,8],lapack:0,laptop:5,larg:[0,1,2,3,5],larger:[0,1,2,3],lasso:[0,1,3,5],last:[0,1,2,3],later:[0,2,3,5],latex:0,latter:[0,3],law:0,layer:0,lbfg:3,lbl:1,lcc:1,ldot:[0,1],lead:[0,1,2,3],lear:3,learn:[1,8],learning_rate_init:0,learning_schedul:3,least:[0,1,2,3,5],leav:[0,1],lectur:[0,1,2,3,5,6],left:[0,1,2,3],legend:[0,1,3],len:[0,1,2],length:[0,3,5],leq:[0,2,3],less:[0,1,2],let:[0,1,2,3],letter:0,level:[0,1,5,6],lib:1,librari:8,licens:[0,5],lie:1,life:0,light:0,like:[0,1,2,3,5],likelihood:0,limit:[0,1],lin_model:0,linalg:[0,2,3],line:[0,1,3],linear:[1,2,3,5],linear_model:[0,1,3],linear_regress:1,linearli:2,linearloc:3,linearregress:[0,1,3],linewidth:[0,1],link:[0,3,5,7],linpack:0,linreg:0,linspac:[0,1],linux:[0,5],liquid:0,list:[0,5],literatur:3,lle:0,lmb:1,lmbd:0,lmbd_val:0,lmbda:3,load:[0,3],load_boston:0,load_breast_canc:3,loc:[0,1,3],local:[0,3],log10:1,log:[0,1,3],log_:0,logarithm:[0,3],logic:0,logist:[0,5],logisticregress:3,logit:3,logreg:3,logspac:[0,1],longer:0,loocv:1,look:[0,1,2,3],loop:[0,1,5],loss:[0,1,2],lot:[0,1],low:[0,1],lower:0,lowercas:0,lowest:3,lstat:0,lstsq:0,m_h:0,m_n:0,m_p:0,machin:[1,8],machinelearn:6,mackai:8,made:[0,1,2,3],mae:0,magnitud:3,mai:[0,1,2,3,5,7],mail:6,main:[0,2,3,8],mainli:[0,1,3,5],maintain:1,major:[0,1,3],make:[0,1,2,3,5,8],make_pipelin:[0,1],makedir:[0,1,3],makeplot:0,malign:3,manag:[0,5],mani:[0,1,3,4,5,8],map:[0,1,3],margin:0,marit:0,mark:0,marker:[0,3],markov:5,mass:[0,2],massag:0,masses2016:0,masses2016ol:0,masses2016tre:0,masseval2016:0,master:6,mat1100:5,mat1110:5,mat1120:5,mat3155:6,mat4155:6,mat:5,match:3,materi:[0,2,3],math:[0,2,3,8],mathbb:[0,1,2,3],mathbf:[0,1,2,3],mathcal:[1,3],mathemat:[0,2,3,5,8],mathemati:0,mathemt:2,mathrm:[0,1,2,3],matmul:2,matnat:8,matplotlib:[0,1,3,5],matric:[2,3,5],matrix:[1,2],matter:3,max:[0,3],max_depth:0,max_it:[0,3],maxdegre:1,maxim:3,maximum:[0,3],maxpolydegre:1,mbox:[1,2],mean:[0,1,2,3],mean_absolute_error:0,mean_squared_error:[0,1,3],mean_squared_log_error:0,meaning:3,meansquarederror:0,meant:3,measur:[0,1],median:0,medv:0,meet:7,mehta:[0,2],mention:[0,3],meshgrid:[0,2],met:0,method:[0,2,5,8],metric:[0,1,3],metropoli:5,mev:0,mglearn:[0,5],mgrid:3,might:[0,3],million:0,min:[0,2],min_:[0,2],mind:[0,3],mine:[0,5],mini:3,minibatch:3,minibathc:3,minim:[0,1,2,3],minima:[0,3],minimum:[0,1,3],minmaxscal:0,minu:3,miss:0,mit:8,mix:5,mkdir:[0,1,3],mlab:1,mle:3,mlpregressor:0,mode:6,model:[1,2,3,5,8],model_select:[0,1,3],modern:[0,1,3,5],modifi:[0,2,3],modul:[0,1,3],moe:2,moment:1,mont:[0,1,5,8],more:[1,2,5],moreov:0,morten:7,most:[0,1,2,3,5,6],motion:0,move:[0,3],mpl:[0,3],mpl_toolkit:3,mplot3d:3,mse:[0,1,2],msle:0,much:[0,1,3],multi:[0,3,5],multiclass:3,multidimension:0,multinomi:3,multipl:[1,3],multipli:[2,3],multitud:0,multivari:[0,5],murphi:8,must:[1,3],mutat:3,mutual:[1,3],myriad:[0,5],n_boostrap:1,n_bootstrap:1,n_epoch:3,n_hidden_neuron:0,n_sampl:1,n_split:1,nabla:3,nabla_:3,naimi:0,naiv:3,name:[0,1,3,5,7],nativ:[0,5],natur:[0,3,8],nbconvert:0,nearli:3,neat:0,neccesari:1,necessarili:5,neck:3,need:[1,2,3],neg:[0,1,3],neg_mean_squared_error:1,neq:3,netlib:0,network:[0,5,8],neural:[0,3,5,8],neural_network:0,neutral:0,neutron:0,never:1,new_hobbit:0,new_sig:1,newaxi:[0,1],next:[0,3],next_guess:3,nice:0,niter:3,nitric:0,nlambda:1,nm_n:0,nmse:1,nois:[0,1,3],noisi:1,non:[0,1,2,3],none:[0,1,3],nonlinear:1,nonneg:[1,3],nonparametr:1,nonsingular:0,nonumb:3,norm:[0,1,2,3],normal:[0,1,2,3,5],normali:0,normpdf:1,notat:[0,1],note:[0,1,2,3],notebook:[0,5],noth:2,notic:0,novel:1,now:[0,1,2,3],nowadai:[0,5],nox:0,nsampl:1,nuclear:2,nuclei:0,nucleon:0,nucleu:0,number:[1,2,3,6,7,8],numberid:3,numer:[0,1,2,3,5,8],numpi:[1,2,3,5],obei:3,object:[0,1],observ:[1,2,3],obtain:[0,1,2,3],obviou:2,obviouli:0,obvious:0,occupi:0,occur:0,odd:[0,3],off:1,offend:0,offer:[0,1,5,6],offic:7,offici:6,often:[0,1,2,3,5],ofter:0,old:3,omit:0,onc:1,one:[0,1,2,5],ones:[0,2,3],onli:[0,1,2,3],onlin:[0,6],open:[0,1,3,5,6],oper:[0,1,2,5],opmiz:3,opportun:0,opt:[0,1],optim:[0,1,2],orang:0,order:[0,1,2,3],ordinari:[1,2,3,5],oreilli:8,org:[0,5],organ:[1,3],orient:1,origin:[0,1],orthogn:2,orthogon:[0,2,3],orthonorm:2,oslo:[5,6,7],osx:[0,5],other:[0,1,2,3,5,6,8],otherwis:[0,3],ouput:3,our:[1,2],ourselv:[0,2,3],out:[0,1,3,5],outcom:[0,3],outlier:0,outlin:1,output:[0,3],over:[0,1,3],overdetermin:0,overfit:1,overlap:3,overlin:[0,1,2],overst:0,overview:8,own:[0,3,5],owner:0,oxid:0,pack:0,packag:[1,2,3,5],page:[0,5],painless:0,pair:[0,5],panda:[1,2,3,5],panel:0,paradigm:0,parallel:0,paramet:[0,1,2,3],parameter:0,parametr:[0,1],part:[0,1,2,6,8],partial:[0,2,3],particip:[5,6],particl:0,particular:[0,1,2,3,8],particularli:[1,2,3],patch:1,path:[0,1,3,5],patient:3,pattern:[0,8],pauli:0,pca:[0,3,5],pdf:[0,1],pedagog:0,penalti:[1,3],pentagon:3,peopl:[0,5],per:[0,1,6],percentag:0,perceptron:[0,3],peregrin:0,perfect:0,perfectli:1,perform:[0,1,3,5,7],perhap:[0,2,3],person:[3,7],perspect:8,peter:8,philosophi:3,phone:7,physic:[0,3,7,8],pick:3,pictur:0,pie:[0,5],pillow:[0,5],pip3:0,pip:[0,5],pipelin:[0,1],pippin:0,place:[0,1,3],plai:[0,1,5],plain:3,plan:[1,7],platform:[0,5],plot:[0,1,3,5],plot_confusion_matrix:3,plot_cumulative_gain:3,plot_roc:3,plot_surfac:3,plt:[0,1,3],plu:[0,3],png:[0,1,3],point:[0,1,3,7],poisson:5,poli:1,poly3:0,poly3_plot:0,polydegre:1,polygon:3,polynomi:[0,1,2,3],polynomial_featur:1,polynomialfeatur:[0,1],polytrop:[0,1],poor:3,popul:0,popular:[0,1,3,5],popularli:0,posit:[0,2,3],possibl:[0,1,3,5,7],postpon:0,potenti:0,power:[0,1,2],practic:[0,1,3],practition:0,precis:[0,2],pred:1,predict:[0,1,3,5,8],predict_proba:3,predictor:[0,2,3],prefer:[0,5],prepar:0,preprocess:[1,3],prerequisit:0,present:0,press:[3,8],pretti:[0,5],previou:[0,2,3],price:0,primari:3,princip:[0,3,5],principl:[0,1,3],print:[0,1,2,3],printout:0,prior:[0,1],privat:0,probabilist:[0,8],probabl:[0,1,3,5],problem:[0,1,2,5],proce:[0,3],procedur:[1,2,3],process:[0,1,3,5,8],prod_:3,produc:[0,2,5],product:[0,1,3,5],profess:0,program:[0,2,5,6],prohibit:1,project:[0,3,5,6,7],project_root_dir:[0,1,3],pronounc:[0,5],proof:[0,3],prop:1,proper:[0,1],properti:[0,1,2,3],proport:0,propto:3,proton:0,prove:3,provid:[0,1,2,3,5,8],psycholog:0,punish:0,purpos:0,pycod:0,pydata:5,pyhton2:0,pylab:[0,3],pypi:5,pyplot:[0,1,3],python3:[0,1,5],pytorch:[0,5],qquad:0,quad:3,quadrat:[0,3],qualiti:[0,5],quantit:1,quantiti:[0,1,2,3],quartil:0,question:[0,1,3],quickli:3,quit:1,r2_score:0,r2score:0,rad:0,radial:0,radiu:0,rais:1,rand:[0,1,3],randint:[1,3],randn:[0,1,3],random:[0,1,2,3,5],random_index:3,random_st:[0,3],randomli:[1,3],rang:[0,1,2,3],rangl:0,rank:2,rapidli:0,rate:[0,3],rather:[0,1,2,3,5],ratio:3,rational:0,ravel:[0,1,2,3],rcond:0,rcparam:[0,3],reach:[1,3],read:[0,1,3],read_csv:[0,1,3],read_fwf:0,reader:0,readi:0,real:[0,1,2,3],realli:0,reason:[0,3,8],recal:[0,1,2,3],recent:[0,1,3],recip:[0,3],recogn:0,recognit:[0,8],recommend:[0,2,5,8],record:6,rectangl:[1,3],recur:[0,5],recurr:5,recurs:[0,5],red:[0,1],redefin:0,reduc:[2,3],reduct:[0,5],refer:[0,1,2,3],refit:1,reflect:0,regard:3,regr_1:0,regr_2:0,regr_3:0,regress:[1,5],regressor:[0,3],regular:[0,3],reilli:8,reinforc:[0,5],rel:[0,1,3],relat:[0,3],relationship:0,relativeerror:0,releas:[0,5],relev:[0,2,3,5],reli:0,reliabl:3,remain:1,rememb:0,remind:[0,2],remov:[0,2],render:0,reorder:3,reorgan:0,repeat:[0,1,3],repeatedli:1,repetit:1,rephras:3,replac:[0,1,3],replica:1,repositori:0,repres:[0,1,3],represent:1,reproduc:[0,5],repuls:0,request:0,requir:[0,1,2,3],resampl:[0,3,5],rescal:0,research:[0,5,8],resembl:1,reserv:1,reshap:[0,1],residenti:0,residu:[0,3],respect:[0,1,2,3],respons:[0,3],rest:[0,2],restat:0,result:[0,1,2,3,5],ret:1,retail:0,retain:[1,2],reward:0,rewrit:[0,1,2,3],rewritten:[1,2],rewrot:3,rho:0,rich:0,ridg:[0,1,5],right:[0,1,2,3],rightarrow:[0,2,3],rigor:0,rise:0,risk:[0,3],river:0,rmse:0,robert:8,robust:0,robustscal:0,role:[0,1,2,5],room:[0,7],root:[0,3],rot:0,round:[0,3],routin:[0,3],row:[0,1,2],rrr:2,rug:3,rule:[0,7],run:[0,1,2,3,5],runtim:1,rust:[0,5],rwidth:1,s_i:3,saddl:3,sai:[0,1,2,3,5],said:[1,3],sake:[0,2,3],sale:0,sam:0,same:[0,1,2,3],sampl:[0,1,2,3,5],samwis:0,sanitize_sequ:1,satisfactori:0,satisfi:[1,3],satur:1,save:[0,1,3],save_fig:[0,1,3],savefig:[0,1,3],scalar:1,scale:[0,2,3,5,7],scaler:[0,3],scan:3,scatter:[0,1,3,5],scenario:3,scheme:3,scienc:[0,3,5,6,8],scientif:[0,5],scientist:0,scikit:[5,8],scikitlearn:0,scikitplot:3,scipi:[0,2,3,5],score:[0,1,3,7],scores_kfold:1,sdg:3,seaborn:[0,3],seamless:[0,5],search:[0,3],sec:1,second:[0,1,3,5],section:4,sector:0,see:[0,1,2,3,7],seed:[0,1,3],seemingli:0,seen:0,segment:3,seldomli:0,select:[0,1,2,6,8],self:[1,8],semest:[3,6,7],semi:3,send:7,senior:6,sens:1,sensit:[0,1],separ:[0,1,5],sequenc:[0,3,5],seri:[0,3],serif:[0,3],serv:[0,3,8],session:[5,6],set:[0,1,2,3,5],set_:1,set_label:1,set_titl:[0,3],set_xlabel:[0,3],set_xlim:3,set_ylabel:[0,3],set_ylim:3,set_ytick:3,setminu:1,setp:1,setup:[0,5],sever:[0,1,2,3,5],sgdreg:3,sgdregressor:3,shape:[0,1,2,3],share:0,she:3,shire:0,shortcom:3,shorthand:0,shortli:0,should:[0,1,2],show:[0,1,2,3],shown:3,shrink:2,shrinkag:2,shuffl:1,side:[0,3],sigh:[0,5],sigma:[0,1,2,3],sigma_1:2,sigma_2:2,sigma_:[0,2],sigma_fn:3,sigma_i:[0,2],sigma_j:2,sigmoid:3,sign:3,significantli:3,sim:1,similar:[0,1,3,5],similarli:[0,2],simpl:[1,2,5],simpler:[0,5],simplest:0,simpli:[0,1,2,5],simplic:[2,3],simplifi:[0,1,5],simul:1,simultan:1,sin:0,sinc:[0,1,2,3,8],singl:[0,3],singular:[0,1,3],site:[0,1,6],situat:[0,2,3],size:[0,1,3],skill:0,skl:0,sklearn:[0,1,3],skplt:3,slice:0,slide:0,slight:1,slightli:[1,2],slow:[0,3],slower:[0,2],small:[0,1,2,3,5],smaller:[0,1,3],smallest:0,smart:1,smooth:[0,3],sneak:0,sns:[0,3],soar:1,social:[0,6],soft:3,softmax:3,softwar:5,sole:0,solid:[0,3],solut:[0,1,2,3],solv:[0,2],solver:3,some:[1,2],someth:3,sometim:0,sophist:0,sopt:3,sort:[0,1,2],sourc:[0,1,5],space:[0,2,3],span:[0,2],spars:0,sparse_mtx:0,special:[0,1,3],specif:[0,1,2,3,5],specifi:[0,1,3],specifici:0,speech:0,sphere:0,spite:0,split:1,spread:0,springer:8,sqrt:[0,1,2,3],squar:[0,1,2,3,5],stabl:[0,2,5],stack:1,stage:3,stai:0,stand:[0,2],standard:[0,1,2],standardscal:[0,3],stanford:3,start:[0,1,3],stat:1,state:[2,3,5],statement:[0,3],statis:2,statist:[0,2,3,8],statu:[0,3],std:[0,1],steep:3,step:0,step_fn:3,step_length:3,still:[1,2,3],stk2100:8,stk4021:8,stk4051:8,stk5000:8,stk:8,stochast:1,stone:[0,3],storag:2,store:[0,3],straight:[0,1,3],straightforward:[0,1,3],stratifi:1,strength:[0,2],strict:3,strictli:3,stroke:3,strongli:[0,5],stronli:0,structur:[0,1,5],stuck:3,student:[0,6,8],studi:[0,3,5,8],studier:8,style:[0,3],subdivid:0,subfield:0,subplot:[0,1,3],subprogram:0,subroutin:0,subsequ:1,subset:[0,1,3,5],subspac:0,substitut:1,subtract:[1,2],succeed:0,success:3,sudo:[0,5],suffer:[0,2],suffici:[1,3],suggest:3,suitabl:0,sum:[0,1,2,3],sum_:[0,1,2,3],sum_i:[1,2,3],sum_k:0,summar:1,summari:6,summat:2,supervis:[0,1,3,5],supplement:3,support:[0,5],suppos:[0,1,2,3],surfac:0,surpris:0,surround:5,survei:0,svd:[0,1],svdinv:2,symbol:[0,5],symmetr:[0,2,3],sympi:[0,5],syntax:0,syntaxerror:0,sys:3,system:[0,3,5,8],systemat:1,t_0:3,t_1:3,tabl:[0,7],tabul:0,tabular:0,tag:[2,3],taht:0,tailor:0,taiwan:0,take:[0,1,2,3,5],taken:[0,1],tangent:3,tanh:3,target:[0,3],task:[0,1],taught:5,tax:0,taylor:3,taylornr:3,teaser:0,techniqu:[0,5,8],technolog:0,tek5040:8,telephon:7,tell:[1,3],temperatur:0,ten:0,tend:1,tendenc:0,tension:1,tensorflow:[0,5,8],term1:[0,2],term2:[0,2],term3:[0,2],term4:[0,2],term:[0,1,2,3],termin:2,test:[1,2,3],test_ind:1,test_scor:3,test_siz:[0,1],testerror:1,text:[0,3,8],than:[0,1,2,5],theano:[0,5],thei:[0,1,2,3],them:[0,3],theme:0,themselv:0,thenc:1,theorem:[1,2,3],theoret:0,theori:[0,3,5,8],thereaft:[0,1,2],therebi:[0,3],therefor:[0,1,3],thereof:[0,1,3],theta:[1,3],theta_linreg:3,thi:[0,1,2,3,4,5,6,8],thing:[0,3],think:[0,1,3],third:[0,3],thirti:3,thorughout:5,those:[0,2,6],thought:1,thousand:0,three:[0,1,5,6,7],threshold:3,through:[0,2,3,5],throughout:[0,5],thu:[0,1,2,3,7],thumb:[0,7],tibshirani:8,ticker:3,tight_layout:3,tild:[0,1,2],till:[0,3],time:[0,1,2,3],tip:4,titl:[0,1,3],to_numer:[0,1],togeth:0,too:[0,1,2,3,8],took:0,tool:[0,1,5],top:[0,1,5],topic:[0,3,5,8],total:[0,1,3,7],toward:3,town:0,traceback:[1,3],track:3,tract:0,tractabl:0,trade:1,tradeoff:[0,2],tradit:[0,1],train:[1,3],train_accuraci:0,train_ind:1,train_test_split:[0,1,3],trainingerror:1,trait:0,transform:[0,1,2,3,5],transpos:2,treat:[0,1,3],tree:[0,5],trevor:8,trial:[0,1,3],triangl:3,triangular:0,tridiagon:0,trillion:5,trivial:0,troubl:0,true_fun:1,tumor:3,tumour:3,tune:0,turn:[0,1,2,3],twice:3,two:[0,1,2,3,6],tx_1:3,type:[0,1,3],typic:[0,2,3],ubuntu:[0,5],uci:0,uio:[7,8],unari:0,unbalanc:1,unbias:[0,1],uncertainti:0,undefin:2,under:[0,1,3,5],underdetermin:0,underfit:1,undergradu:6,underli:[0,3],understand:[0,3,5],unexpect:1,uniform:[0,2,3],uniformli:3,unimport:3,union:1,uniqu:[0,1,3],unit:0,unitari:[0,2],unitarili:0,univers:[3,5,6,7],unknow:0,unknown:[0,1],unless:[0,1,3],unlik:3,unseen:3,unsupervis:[0,5],unsymmetr:0,until:3,updat:1,upload:5,upper:0,uppercas:0,ups:0,usag:[0,5],usd10000:0,usd:0,use:[0,1,2,3,5],usecol:0,used:[0,1,2,5,8],useful:[0,1,3,5,8],user:[0,3,5],uses:[0,1,2],using:[1,2],usual:[0,3],util:[0,1,3],valid:[0,3,5],valu:[0,1,3,5],van:[0,2],vandenbergh:3,vandermond:0,vanish:3,varepsilon:1,varepsilon_:1,varepsilon_i:1,vari:[0,1],variabl:[0,1],varianc:[0,2,3,5],variance_i:2,variance_x:2,variant:[0,1,3],varieti:[0,5],variou:[0,2,3,5],varou:5,vdot:3,vec:1,vector:[1,2,3,5],ventur:[0,5],veri:[0,1,3,8],verifi:0,versatil:0,version:[0,5],vert:[0,2,3],vert_1:2,vert_2:2,vertic:1,via:[0,1,2,3,5,6,7,8],video:[0,1,2,3,5,6],view:8,viridi:0,vision:0,visual:[0,5],volum:0,vstack:[0,2],wai:[0,1,2,3,4],wang:0,want:[0,1,3,5],warrant:1,web:[0,5,6],websit:[0,6],week:[3,6],weekli:6,weight:[0,1,3],welcom:5,well:[0,1,2,3,5,8],were:[0,1,2,3],wessel:[0,2],what:[1,2,3,5],when:[0,1,2],where:[0,1,2,3,5,7],wherea:1,whether:[0,3],which:[0,1,2,3,5,6,7],who:[0,6],whose:1,whow:2,why:[0,3],wide:[0,1,3,5],widehat:1,width:0,wieringen:[0,2],wing:7,wiscons:3,wise:[0,2],wish:[0,3],within:[0,3,8],without:[0,2,3],won:0,word:0,work:[0,1,3,5,6],world:0,worldwid:0,wors:[0,1],would:[0,1,3],wrap:0,write:[0,3,4],written:[0,2,3,5],wrote:2,www:[0,5,8],x_0:[0,2],x_1:[0,1,2,3],x_2:[0,1,2,3],x_i:[0,1,2,3],x_ix_:0,x_n:[1,3],x_p:3,x_test:[0,1,3],x_test_scal:[0,3],x_train:[0,1,3],x_train_scal:[0,3],xarrai:[0,5],xbnew:3,xcode:[0,5],xlabel:[0,1,3],xlim:1,xmesh:3,xnew:[0,3],xpd:2,xplot:0,xt_x:3,xtest:1,xtick:1,xtrain:1,y_0:[0,2],y_1:[0,2,3],y_2:[0,2],y_3:0,y_data:0,y_i:[0,1,2,3],y_ix_:0,y_ix_i:3,y_j:1,y_model:0,y_n:3,y_pred:[0,1,3],y_proba:3,y_test:[0,1,3],y_test_predict:0,y_train:[0,1,3],y_train_predict:0,year:[0,5],yes:[1,3],yet:[0,1],yield:[0,1,3],ylabel:[0,1,3],ylim:1,ymesh:3,you:[0,1,3,5,8],young:0,your:[0,1,3,5,8],yourself:[0,3],youtub:5,ypred:1,ypredict2:3,ypredict:[0,3],yridg:0,ytest:1,ytick:1,ytild:[0,1],ytildenp:0,ytrain:1,z_0:0,z_1:0,z_2:0,zero:[0,1,2,3],zm_h:0,zone:0,zoom:7},titles:["1. Linear Regression, basic Elements","2. Resampling Methods","3. Ridge and Lasso Regression","4. Logistic Regression","Content in Jupyter Book","Applied Data Analysis and Machine Learning","Teaching schedule with links to material","Teachers and Grading","Textbooks"],titleterms:{"case":3,"final":3,"function":[0,2,3],"import":0,And:3,The:[0,1,2,3,5],Useful:[0,5],Using:3,algorithm:2,algortithm:3,analysi:5,ani:3,appli:5,approach:3,arrai:0,basic:[0,3],better:2,bia:1,book:4,bootstrap:1,boston:0,brief:3,cancer:3,central:[3,5],chi:0,code:[0,3],comput:3,condit:3,conjug:3,content:4,convex:3,correl:2,correspond:3,cost:3,cours:[5,8],covari:2,cross:1,cython:0,data:[0,3,5],decomposit:2,degre:0,dens:0,deriv:3,descent:3,differ:3,economi:2,element:0,equat:[0,3],etc:0,exampl:[0,3],express:3,extend:3,famou:0,fantast:2,featur:0,first:3,fit:0,frank:0,freedom:0,fridai:3,geometr:3,grade:7,gradient:3,handl:0,has:5,hessian:3,homework:3,hous:0,ideal:3,inform:7,instal:[0,5],instructor:7,interpret:3,introduc:2,introduct:[0,1,5],iter:3,julia:0,jupyt:4,lasso:2,learn:[0,3,5],librari:[0,5],likelihood:3,limit:3,linear:0,link:[2,6,8],logist:3,loss:3,machin:[0,3,5],materi:6,matric:0,matrix:[0,3],matter:0,meet:0,method:[1,3],model:0,more:[0,3],need:0,network:3,newton:3,nuclear:0,nueral:3,numba:0,number:0,numpi:0,one:3,optim:[3,5],organ:0,oslo:8,our:[0,3],outcom:5,overarch:0,packag:0,panda:0,part:[3,5],preprocess:0,prerequisit:5,problem:3,program:3,python:[0,5],raphson:3,reduc:0,regress:[0,2,3],regular:2,relev:8,remind:[1,3],requir:5,resampl:1,revisit:3,ridg:[2,3],schedul:6,scikit:[0,3],sensit:3,septemb:3,sgd:3,simpl:[0,3],singular:2,size:2,slightli:3,softwar:0,solv:3,some:[0,3],split:0,standard:3,state:0,statist:[1,5],steepest:3,step:[1,3],stochast:3,stop:3,svd:2,teach:6,teacher:7,technolog:5,test:0,textbook:8,than:3,tradeoff:1,train:0,two:5,understand:2,univers:8,used:3,using:[0,3],valid:1,valu:2,variabl:3,varianc:1,variou:1,vector:0,view:0,what:0,when:3,wisconsin:3}}) \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/teachers.html b/doc/LectureNotes/_build/html/teachers.html new file mode 100644 index 000000000..88376df83 --- /dev/null +++ b/doc/LectureNotes/_build/html/teachers.html @@ -0,0 +1,320 @@ + + + + + + + + Teachers and Grading — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ +
+
+ +
+ + + + + + + + + + + + + + +
+ + +
+ +
+ + Contents +
+ + +
+
+
+
+
+ +
+ +
+

Teachers and Grading

+
+

Instructor information

+
    +
  • Name: Morten Hjorth-Jensen

  • +
  • Email: morten.hjorth-jensen@fys.uio.no

  • +
  • Phone: +47-48257387

  • +
  • Office: Department of Physics, University of Oslo, Eastern wing, room FØ470

  • +
  • Office hours: Anytime! In Fall Semester 2020 (FS20), as a rule of thumb office hours are planned via computer or telephone. Individual or group office hours will be performed via zoom. Feel free to send an email for planning. In person meetings may also be possible if allowed by the University of Oslo’s COVID-19 instructions (see below for links).

  • +
+
+
+

Grading

+

Grading scale: Grades are awarded on a scale from A to F, where A is the best grade and F is a fail. There are three projects which are graded and each project counts 1/3 of the final grade. The total score is thus the average from all three projects.

+

The final number of points is based on the average of all projects (including eventual additional points) and the grade follows the following table:

+
    +
  • 92-100 points: A

  • +
  • 77-91 points: B

  • +
  • 58-76 points: C

  • +
  • 46-57 points: D

  • +
  • 40-45 points: E

  • +
  • 0-39 points: F-failed

  • +
+
+
+ + + + +
+ + + + +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/html/textbooks.html b/doc/LectureNotes/_build/html/textbooks.html new file mode 100644 index 000000000..930db5797 --- /dev/null +++ b/doc/LectureNotes/_build/html/textbooks.html @@ -0,0 +1,328 @@ + + + + + + + + Textbooks — Applied Data Analysis and Machine Learning + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+ + + + + + + + +
+ +
+
+ +
+ + + + + + + + + + + + + + +
+ + +
+ +
+ + Contents +
+ + +
+
+
+
+
+ +
+ +
+

Textbooks

+

Recommended textbooks:

+ +

The books by Bishop and Hastie et al. can be downloaded for free if you access the university library via an IP number of your home university.

+

General learning book on statistical analysis:

+
    +
  • Christian Robert and George Casella, Monte Carlo Statistical Methods, Springer

  • +
  • Peter Hoff, A first course in Bayesian statistical models, Springer

  • +
+

General Machine Learning Books:

+
    +
  • Kevin Murphy, Machine Learning: A Probabilistic Perspective, MIT Press

  • +
  • Christopher M. Bishop, Pattern Recognition and Machine Learning, Springer

  • +
  • David J.C. MacKay, Information Theory, Inference, and Learning Algorithms, Cambridge University Press

  • +
  • David Barber, Bayesian Reasoning and Machine Learning, Cambridge University Press

  • +
+ +
+ + + + +
+ + + + +
+
+
+
+

+ + By Morten Hjorth-Jensen
+ + © Copyright 2020.
+

+
+
+
+ + +
+
+ + + + + + + + \ No newline at end of file diff --git a/doc/LectureNotes/_build/jupyter_execute/chapter1.ipynb b/doc/LectureNotes/_build/jupyter_execute/chapter1.ipynb new file mode 100644 index 000000000..964471af9 --- /dev/null +++ b/doc/LectureNotes/_build/jupyter_execute/chapter1.ipynb @@ -0,0 +1,4129 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Linear Regression, basic Elements\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK3155/h20/forelesningsvideoer/LectureAug21.mp4?vrtx=view-as-webpage)\n", + "\n", + "\n", + "## Introduction\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "Our emphasis throughout this series of lectures \n", + "is on understanding the mathematical aspects of\n", + "different algorithms used in the fields of data analysis and machine learning. \n", + "\n", + "However, where possible we will emphasize the\n", + "importance of using available software. We start thus with a hands-on\n", + "and top-down approach to machine learning. The aim is thus to start with\n", + "relevant data or data we have produced \n", + "and use these to introduce statistical data analysis\n", + "concepts and machine learning algorithms before we delve into the\n", + "algorithms themselves. The examples we will use in the beginning, start with simple\n", + "polynomials with random noise added. We will use the Python\n", + "software package [Scikit-Learn](http://scikit-learn.org/stable/) and\n", + "introduce various machine learning algorithms to make fits of\n", + "the data and predictions. We move thereafter to more interesting\n", + "cases such as data from say experiments (below we will look at experimental nuclear binding energies as an example).\n", + "These are examples where we can easily set up the data and\n", + "then use machine learning algorithms included in for example\n", + "**Scikit-Learn**. \n", + "\n", + "These examples will serve us the purpose of getting\n", + "started. Furthermore, they allow us to catch more than two birds with\n", + "a stone. They will allow us to bring in some programming specific\n", + "topics and tools as well as showing the power of various Python \n", + "libraries for machine learning and statistical data analysis. \n", + "\n", + "Here, we will mainly focus on two\n", + "specific Python packages for Machine Learning, Scikit-Learn and\n", + "Tensorflow (see below for links etc). Moreover, the examples we\n", + "introduce will serve as inputs to many of our discussions later, as\n", + "well as allowing you to set up models and produce your own data and\n", + "get started with programming.\n", + "\n", + "\n", + "\n", + "## What is Machine Learning?\n", + "\n", + "Statistics, data science and machine learning form important fields of\n", + "research in modern science. They describe how to learn and make\n", + "predictions from data, as well as allowing us to extract important\n", + "correlations about physical process and the underlying laws of motion\n", + "in large data sets. The latter, big data sets, appear frequently in\n", + "essentially all disciplines, from the traditional Science, Technology,\n", + "Mathematics and Engineering fields to Life Science, Law, education\n", + "research, the Humanities and the Social Sciences. \n", + "\n", + "It has become more\n", + "and more common to see research projects on big data in for example\n", + "the Social Sciences where extracting patterns from complicated survey\n", + "data is one of many research directions. Having a solid grasp of data\n", + "analysis and machine learning is thus becoming central to scientific\n", + "computing in many fields, and competences and skills within the fields\n", + "of machine learning and scientific computing are nowadays strongly\n", + "requested by many potential employers. The latter cannot be\n", + "overstated, familiarity with machine learning has almost become a\n", + "prerequisite for many of the most exciting employment opportunities,\n", + "whether they are in bioinformatics, life science, physics or finance,\n", + "in the private or the public sector. This author has had several\n", + "students or met students who have been hired recently based on their\n", + "skills and competences in scientific computing and data science, often\n", + "with marginal knowledge of machine learning.\n", + "\n", + "Machine learning is a subfield of computer science, and is closely\n", + "related to computational statistics. It evolved from the study of\n", + "pattern recognition in artificial intelligence (AI) research, and has\n", + "made contributions to AI tasks like computer vision, natural language\n", + "processing and speech recognition. Many of the methods we will study are also \n", + "strongly rooted in basic mathematics and physics research. \n", + "\n", + "Ideally, machine learning represents the science of giving computers\n", + "the ability to learn without being explicitly programmed. The idea is\n", + "that there exist generic algorithms which can be used to find patterns\n", + "in a broad class of data sets without having to write code\n", + "specifically for each problem. The algorithm will build its own logic\n", + "based on the data. You should however always keep in mind that\n", + "machines and algorithms are to a large extent developed by humans. The\n", + "insights and knowledge we have about a specific system, play a central\n", + "role when we develop a specific machine learning algorithm. \n", + "\n", + "Machine learning is an extremely rich field, in spite of its young\n", + "age. The increases we have seen during the last three decades in\n", + "computational capabilities have been followed by developments of\n", + "methods and techniques for analyzing and handling large date sets,\n", + "relying heavily on statistics, computer science and mathematics. The\n", + "field is rather new and developing rapidly. Popular software packages\n", + "written in Python for machine learning like\n", + "[Scikit-learn](http://scikit-learn.org/stable/),\n", + "[Tensorflow](https://www.tensorflow.org/),\n", + "[PyTorch](http://pytorch.org/) and [Keras](https://keras.io/), all\n", + "freely available at their respective GitHub sites, encompass\n", + "communities of developers in the thousands or more. And the number of\n", + "code developers and contributors keeps increasing. Not all the\n", + "algorithms and methods can be given a rigorous mathematical\n", + "justification, opening up thereby large rooms for experimenting and\n", + "trial and error and thereby exciting new developments. However, a\n", + "solid command of linear algebra, multivariate theory, probability\n", + "theory, statistical data analysis, understanding errors and Monte\n", + "Carlo methods are central elements in a proper understanding of many\n", + "of algorithms and methods we will discuss.\n", + "\n", + "\n", + "\n", + "The approaches to machine learning are many, but are often split into\n", + "two main categories. In *supervised learning* we know the answer to a\n", + "problem, and let the computer deduce the logic behind it. On the other\n", + "hand, *unsupervised learning* is a method for finding patterns and\n", + "relationship in data sets without any prior knowledge of the system.\n", + "Some authours also operate with a third category, namely\n", + "*reinforcement learning*. This is a paradigm of learning inspired by\n", + "behavioral psychology, where learning is achieved by trial-and-error,\n", + "solely from rewards and punishment.\n", + "\n", + "Another way to categorize machine learning tasks is to consider the\n", + "desired output of a system. Some of the most common tasks are:\n", + "\n", + " * Classification: Outputs are divided into two or more classes. The goal is to produce a model that assigns inputs into one of these classes. An example is to identify digits based on pictures of hand-written ones. Classification is typically supervised learning.\n", + "\n", + " * Regression: Finding a functional relationship between an input data set and a reference data set. The goal is to construct a function that maps input data to continuous output values.\n", + "\n", + " * Clustering: Data are divided into groups with certain common traits, without knowing the different groups beforehand. It is thus a form of unsupervised learning.\n", + "\n", + "The methods we cover have three main topics in common, irrespective of\n", + "whether we deal with supervised or unsupervised learning. The first\n", + "ingredient is normally our data set (which can be subdivided into\n", + "training and test data), the second item is a model which is normally a\n", + "function of some parameters. The model reflects our knowledge of the system (or lack thereof). As an example, if we know that our data show a behavior similar to what would be predicted by a polynomial, fitting our data to a polynomial of some degree would then determin our model. \n", + "\n", + "The last ingredient is a so-called **cost**\n", + "function which allows us to present an estimate on how good our model\n", + "is in reproducing the data it is supposed to train. \n", + "At the heart of basically all ML algorithms there are so-called minimization algorithms, often we end up with various variants of **gradient** methods.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Software and needed installations\n", + "\n", + "We will make extensive use of Python as programming language and its\n", + "myriad of available libraries. You will find\n", + "Jupyter notebooks invaluable in your work. You can run **R**\n", + "codes in the Jupyter/IPython notebooks, with the immediate benefit of\n", + "visualizing your data. You can also use compiled languages like C++,\n", + "Rust, Julia, Fortran etc if you prefer. The focus in these lectures will be\n", + "on Python.\n", + "\n", + "\n", + "If you have Python installed (we strongly recommend Python3) and you feel\n", + "pretty familiar with installing different packages, we recommend that\n", + "you install the following Python packages via **pip** as \n", + "\n", + "1. pip install numpy scipy matplotlib ipython scikit-learn mglearn sympy pandas pillow \n", + "\n", + "For Python3, replace **pip** with **pip3**.\n", + "\n", + "For OSX users we recommend, after having installed Xcode, to\n", + "install **brew**. Brew allows for a seamless installation of additional\n", + "software via for example \n", + "\n", + "1. brew install python3\n", + "\n", + "For Linux users, with its variety of distributions like for example the widely popular Ubuntu distribution,\n", + "you can use **pip** as well and simply install Python as \n", + "\n", + "1. sudo apt-get install python3 (or python for pyhton2.7)\n", + "\n", + "etc etc. \n", + "\n", + "\n", + "\n", + "## Python installers\n", + "\n", + "If you don't want to perform these operations separately and venture\n", + "into the hassle of exploring how to set up dependencies and paths, we\n", + "recommend two widely used distrubutions which set up all relevant\n", + "dependencies for Python, namely \n", + "\n", + "* [Anaconda](https://docs.anaconda.com/), \n", + "\n", + "which is an open source\n", + "distribution of the Python and R programming languages for large-scale\n", + "data processing, predictive analytics, and scientific computing, that\n", + "aims to simplify package management and deployment. Package versions\n", + "are managed by the package management system **conda**. \n", + "\n", + "* [Enthought canopy](https://www.enthought.com/product/canopy/) \n", + "\n", + "is a Python\n", + "distribution for scientific and analytic computing distribution and\n", + "analysis environment, available for free and under a commercial\n", + "license.\n", + "\n", + "Furthermore, [Google's Colab](https://colab.research.google.com/notebooks/welcome.ipynb) is a free Jupyter notebook environment that requires \n", + "no setup and runs entirely in the cloud. Try it out!\n", + "\n", + "\n", + "## Useful Python libraries\n", + "Here we list several useful Python libraries we strongly recommend (if you use anaconda many of these are already there)\n", + "\n", + "* [NumPy](https://www.numpy.org/) is a highly popular library for large, multi-dimensional arrays and matrices, along with a large collection of high-level mathematical functions to operate on these arrays\n", + "\n", + "* [The pandas](https://pandas.pydata.org/) library provides high-performance, easy-to-use data structures and data analysis tools \n", + "\n", + "* [Xarray](http://xarray.pydata.org/en/stable/) is a Python package that makes working with labelled multi-dimensional arrays simple, efficient, and fun!\n", + "\n", + "* [Scipy](https://www.scipy.org/) (pronounced “Sigh Pie”) is a Python-based ecosystem of open-source software for mathematics, science, and engineering. \n", + "\n", + "* [Matplotlib](https://matplotlib.org/) is a Python 2D plotting library which produces publication quality figures in a variety of hardcopy formats and interactive environments across platforms.\n", + "\n", + "* [Autograd](https://github.com/HIPS/autograd) can automatically differentiate native Python and Numpy code. It can handle a large subset of Python's features, including loops, ifs, recursion and closures, and it can even take derivatives of derivatives of derivatives\n", + "\n", + "* [SymPy](https://www.sympy.org/en/index.html) is a Python library for symbolic mathematics. \n", + "\n", + "* [scikit-learn](https://scikit-learn.org/stable/) has simple and efficient tools for machine learning, data mining and data analysis\n", + "\n", + "* [TensorFlow](https://www.tensorflow.org/) is a Python library for fast numerical computing created and released by Google\n", + "\n", + "* [Keras](https://keras.io/) is a high-level neural networks API, written in Python and capable of running on top of TensorFlow, CNTK, or Theano\n", + "\n", + "* And many more such as [pytorch](https://pytorch.org/), [Theano](https://pypi.org/project/Theano/) etc \n", + "\n", + "## Installing R, C++, cython or Julia\n", + "\n", + "You will also find it convenient to utilize **R**. We will mainly\n", + "use Python during our lectures and in various projects and exercises.\n", + "Those of you\n", + "already familiar with **R** should feel free to continue using **R**, keeping\n", + "however an eye on the parallel Python set ups. Similarly, if you are a\n", + "Python afecionado, feel free to explore **R** as well. Jupyter/Ipython\n", + "notebook allows you to run **R** codes interactively in your\n", + "browser. The software library **R** is really tailored for statistical data analysis\n", + "and allows for an easy usage of the tools and algorithms we will discuss in these\n", + "lectures.\n", + "\n", + "To install **R** with Jupyter notebook \n", + "[follow the link here](https://mpacer.org/maths/r-kernel-for-ipython-notebook)\n", + "\n", + "\n", + "\n", + "\n", + "## Installing R, C++, cython, Numba etc\n", + "\n", + "\n", + "For the C++ aficionados, Jupyter/IPython notebook allows you also to\n", + "install C++ and run codes written in this language interactively in\n", + "the browser. Since we will emphasize writing many of the algorithms\n", + "yourself, you can thus opt for either Python or C++ (or Fortran or other compiled languages) as programming\n", + "languages.\n", + "\n", + "To add more entropy, **cython** can also be used when running your\n", + "notebooks. It means that Python with the jupyter notebook\n", + "setup allows you to integrate widely popular softwares and tools for\n", + "scientific computing. Similarly, the \n", + "[Numba Python package](https://numba.pydata.org/) delivers increased performance\n", + "capabilities with minimal rewrites of your codes. With its\n", + "versatility, including symbolic operations, Python offers a unique\n", + "computational environment. Your jupyter notebook can easily be\n", + "converted into a nicely rendered **PDF** file or a Latex file for\n", + "further processing. For example, convert to latex as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + " pycod jupyter nbconvert filename.ipynb --to latex \n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "And to add more versatility, the Python package [SymPy](http://www.sympy.org/en/index.html) is a Python library for symbolic mathematics. It aims to become a full-featured computer algebra system (CAS) and is entirely written in Python. \n", + "\n", + "Finally, if you wish to use the light mark-up language \n", + "[doconce](https://github.com/hplgit/doconce) you can convert a standard ascii text file into various HTML \n", + "formats, ipython notebooks, latex files, pdf files etc with minimal edits. These lectures were generated using **doconce**.\n", + "\n", + "\n", + "\n", + "## Numpy examples and Important Matrix and vector handling packages\n", + "\n", + "There are several central software libraries for linear algebra and eigenvalue problems. Several of the more\n", + "popular ones have been wrapped into ofter software packages like those from the widely used text **Numerical Recipes**. The original source codes in many of the available packages are often taken from the widely used\n", + "software package LAPACK, which follows two other popular packages\n", + "developed in the 1970s, namely EISPACK and LINPACK. We describe them shortly here.\n", + "\n", + " * LINPACK: package for linear equations and least square problems.\n", + "\n", + " * LAPACK:package for solving symmetric, unsymmetric and generalized eigenvalue problems. From LAPACK's website it is possible to download for free all source codes from this library. Both C/C++ and Fortran versions are available.\n", + "\n", + " * BLAS (I, II and III): (Basic Linear Algebra Subprograms) are routines that provide standard building blocks for performing basic vector and matrix operations. Blas I is vector operations, II vector-matrix operations and III matrix-matrix operations. Highly parallelized and efficient codes, all available for download from .\n", + "\n", + "## Basic Matrix Features\n", + "\n", + "Matrix properties reminder" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{A} =\n", + " \\begin{bmatrix} a_{11} & a_{12} & a_{13} & a_{14} \\\\\n", + " a_{21} & a_{22} & a_{23} & a_{24} \\\\\n", + " a_{31} & a_{32} & a_{33} & a_{34} \\\\\n", + " a_{41} & a_{42} & a_{43} & a_{44}\n", + " \\end{bmatrix}\\qquad\n", + "\\mathbf{I} =\n", + " \\begin{bmatrix} 1 & 0 & 0 & 0 \\\\\n", + " 0 & 1 & 0 & 0 \\\\\n", + " 0 & 0 & 1 & 0 \\\\\n", + " 0 & 0 & 0 & 1\n", + " \\end{bmatrix}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The inverse of a matrix is defined by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{A}^{-1} \\cdot \\mathbf{A} = I\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "
Relations Name matrix elements
$A = A^{T}$ symmetric $a_{ij} = a_{ji}$
$A = \\left (A^{T} \\right )^{-1}$ real orthogonal $\\sum_k a_{ik} a_{jk} = \\sum_k a_{ki} a_{kj} = \\delta_{ij}$
$A = A^{ * }$ real matrix $a_{ij} = a_{ij}^{ * }$
$A = A^{\\dagger}$ hermitian $a_{ij} = a_{ji}^{ * }$
$A = \\left (A^{\\dagger} \\right )^{-1}$ unitary $\\sum_k a_{ik} a_{jk}^{ * } = \\sum_k a_{ki}^{ * } a_{kj} = \\delta_{ij}$
\n", + "\n", + "\n", + "### Some famous Matrices\n", + "\n", + " * Diagonal if $a_{ij}=0$ for $i\\ne j$\n", + "\n", + " * Upper triangular if $a_{ij}=0$ for $i > j$\n", + "\n", + " * Lower triangular if $a_{ij}=0$ for $i < j$\n", + "\n", + " * Upper Hessenberg if $a_{ij}=0$ for $i > j+1$\n", + "\n", + " * Lower Hessenberg if $a_{ij}=0$ for $i < j+1$\n", + "\n", + " * Tridiagonal if $a_{ij}=0$ for $|i -j| > 1$\n", + "\n", + " * Lower banded with bandwidth $p$: $a_{ij}=0$ for $i > j+p$\n", + "\n", + " * Upper banded with bandwidth $p$: $a_{ij}=0$ for $i < j+p$\n", + "\n", + " * Banded, block upper triangular, block lower triangular....\n", + "\n", + "### More Basic Matrix Features\n", + "\n", + "Some Equivalent Statements\n", + "For an $N\\times N$ matrix $\\mathbf{A}$ the following properties are all equivalent\n", + "\n", + " * If the inverse of $\\mathbf{A}$ exists, $\\mathbf{A}$ is nonsingular.\n", + "\n", + " * The equation $\\mathbf{Ax}=0$ implies $\\mathbf{x}=0$.\n", + "\n", + " * The rows of $\\mathbf{A}$ form a basis of $R^N$.\n", + "\n", + " * The columns of $\\mathbf{A}$ form a basis of $R^N$.\n", + "\n", + " * $\\mathbf{A}$ is a product of elementary matrices.\n", + "\n", + " * $0$ is not eigenvalue of $\\mathbf{A}$.\n", + "\n", + "## Numpy and arrays\n", + "[Numpy](http://www.numpy.org/) provides an easy way to handle arrays in Python. The standard way to import this library is as" + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here follows a simple example where we set up an array of ten elements, all determined by random numbers drawn according to the normal distribution," + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[ 1.01043351 0.90745348 0.55248701 -0.3588324 0.26881845 0.56326636\n", + " 0.10659197 -0.97644907 0.16328353 -2.30075596]\n" + ] + } + ], + "source": [ + "n = 10\n", + "x = np.random.normal(size=n)\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We defined a vector $x$ with $n=10$ elements with its values given by the Normal distribution $N(0,1)$.\n", + "Another alternative is to declare a vector as follows" + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[1 2 3]\n" + ] + } + ], + "source": [ + "import numpy as np\n", + "x = np.array([1, 2, 3])\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here we have defined a vector with three elements, with $x_0=1$, $x_1=2$ and $x_2=3$. Note that both Python and C++\n", + "start numbering array elements from $0$ and on. This means that a vector with $n$ elements has a sequence of entities $x_0, x_1, x_2, \\dots, x_{n-1}$. We could also let (recommended) Numpy to compute the logarithms of a specific array as" + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[1.38629436 1.94591015 2.07944154]\n" + ] + } + ], + "source": [ + "import numpy as np\n", + "x = np.log(np.array([4, 7, 8]))\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the last example we used Numpy's unary function $np.log$. This function is\n", + "highly tuned to compute array elements since the code is vectorized\n", + "and does not require looping. We normaly recommend that you use the\n", + "Numpy intrinsic functions instead of the corresponding **log** function\n", + "from Python's **math** module. The looping is done explicitely by the\n", + "**np.log** function. The alternative, and slower way to compute the\n", + "logarithms of a vector would be to write" + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[1 1 2]\n" + ] + } + ], + "source": [ + "import numpy as np\n", + "from math import log\n", + "x = np.array([4, 7, 8])\n", + "for i in range(0, len(x)):\n", + " x[i] = log(x[i])\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We note that our code is much longer already and we need to import the **log** function from the **math** module. \n", + "The attentive reader will also notice that the output is $[1, 1, 2]$. Python interprets automagically our numbers as integers (like the **automatic** keyword in C++). To change this we could define our array elements to be double precision numbers as" + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[1.38629436 1.94591015 2.07944154]\n" + ] + } + ], + "source": [ + "import numpy as np\n", + "x = np.log(np.array([4, 7, 8], dtype = np.float64))\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or simply write them as double precision numbers (Python uses 64 bits as default for floating point type variables), that is" + ] + }, + { + "cell_type": "code", + "execution_count": 7, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "ename": "SyntaxError", + "evalue": "invalid syntax (, line 3)", + "output_type": "error", + "traceback": [ + "\u001b[0;36m File \u001b[0;32m\"\"\u001b[0;36m, line \u001b[0;32m3\u001b[0m\n\u001b[0;31m print(x)\u001b[0m\n\u001b[0m ^\u001b[0m\n\u001b[0;31mSyntaxError\u001b[0m\u001b[0;31m:\u001b[0m invalid syntax\n" + ] + } + ], + "source": [ + "import numpy as np\n", + "x = np.log(np.array([4.0, 7.0, 8.0])\n", + "print(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "To check the number of bytes (remember that one byte contains eight bits for double precision variables), you can use simple use the **itemsize** functionality (the array $x$ is actually an object which inherits the functionalities defined in Numpy) as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "x = np.log(np.array([4.0, 7.0, 8.0])\n", + "print(x.itemsize)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Matrices in Python\n", + "\n", + "Having defined vectors, we are now ready to try out matrices. We can\n", + "define a $3 \\times 3 $ real matrix $\\hat{A}$ as (recall that we user\n", + "lowercase letters for vectors and uppercase letters for matrices)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))\n", + "print(A)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we use the **shape** function we would get $(3, 3)$ as output, that is verifying that our matrix is a $3\\times 3$ matrix. We can slice the matrix and print for example the first column (Python organized matrix elements in a row-major order, see below) as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))\n", + "# print the first column, row-major order and elements start with 0\n", + "print(A[:,0])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can continue this was by printing out other columns or rows. The example here prints out the second column" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ]))\n", + "# print the first column, row-major order and elements start with 0\n", + "print(A[1,:])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Numpy contains many other functionalities that allow us to slice, subdivide etc etc arrays. We strongly recommend that you look up the [Numpy website for more details](http://www.numpy.org/). Useful functions when defining a matrix are the **np.zeros** function which declares a matrix of a given dimension and sets all elements to zero" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "n = 10\n", + "# define a matrix of dimension 10 x 10 and set all elements to zero\n", + "A = np.zeros( (n, n) )\n", + "print(A)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or initializing all elements to" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "n = 10\n", + "# define a matrix of dimension 10 x 10 and set all elements to one\n", + "A = np.ones( (n, n) )\n", + "print(A)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or as unitarily distributed random numbers (see the material on random number generators in the statistics part)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "n = 10\n", + "# define a matrix of dimension 10 x 10 and set all elements to random numbers with x \\in [0, 1]\n", + "A = np.random.rand(n, n)\n", + "print(A)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "As we will see throughout these lectures, there are several extremely useful functionalities in Numpy.\n", + "As an example, consider the discussion of the covariance matrix. Suppose we have defined three vectors\n", + "$\\hat{x}, \\hat{y}, \\hat{z}$ with $n$ elements each. The covariance matrix is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\hat{\\Sigma} = \\begin{bmatrix} \\sigma_{xx} & \\sigma_{xy} & \\sigma_{xz} \\\\\n", + " \\sigma_{yx} & \\sigma_{yy} & \\sigma_{yz} \\\\\n", + " \\sigma_{zx} & \\sigma_{zy} & \\sigma_{zz} \n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where for example" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\sigma_{xy} =\\frac{1}{n} \\sum_{i=0}^{n-1}(x_i- \\overline{x})(y_i- \\overline{y}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The Numpy function **np.cov** calculates the covariance elements using the factor $1/(n-1)$ instead of $1/n$ since it assumes we do not have the exact mean values. \n", + "The following simple function uses the **np.vstack** function which takes each vector of dimension $1\\times n$ and produces a $3\\times n$ matrix $\\hat{W}$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\hat{W} = \\begin{bmatrix} x_0 & y_0 & z_0 \\\\\n", + " x_1 & y_1 & z_1 \\\\\n", + " x_2 & y_2 & z_2 \\\\\n", + " \\dots & \\dots & \\dots \\\\\n", + " x_{n-2} & y_{n-2} & z_{n-2} \\\\\n", + " x_{n-1} & y_{n-1} & z_{n-1}\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which in turn is converted into into the $3\\times 3$ covariance matrix\n", + "$\\hat{\\Sigma}$ via the Numpy function **np.cov()**. We note that we can also calculate\n", + "the mean value of each set of samples $\\hat{x}$ etc using the Numpy\n", + "function **np.mean(x)**. We can also extract the eigenvalues of the\n", + "covariance matrix through the **np.linalg.eig()** function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "import numpy as np\n", + "\n", + "n = 100\n", + "x = np.random.normal(size=n)\n", + "print(np.mean(x))\n", + "y = 4+3*x+np.random.normal(size=n)\n", + "print(np.mean(y))\n", + "z = x**3+np.random.normal(size=n)\n", + "print(np.mean(z))\n", + "W = np.vstack((x, y, z))\n", + "Sigma = np.cov(W)\n", + "print(Sigma)\n", + "Eigvals, Eigvecs = np.linalg.eig(Sigma)\n", + "print(Eigvals)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "%matplotlib inline\n", + "\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from scipy import sparse\n", + "eye = np.eye(4)\n", + "print(eye)\n", + "sparse_mtx = sparse.csr_matrix(eye)\n", + "print(sparse_mtx)\n", + "x = np.linspace(-10,10,100)\n", + "y = np.sin(x)\n", + "plt.plot(x,y,marker='x')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Meet the Pandas\n", + "\n", + "\n", + "\n", + "\n", + "Another useful Python package is\n", + "[pandas](https://pandas.pydata.org/), which is an open source library\n", + "providing high-performance, easy-to-use data structures and data\n", + "analysis tools for Python. **pandas** stands for panel data, a term borrowed from econometrics and is an efficient library for data analysis with an emphasis on tabular data.\n", + "**pandas** has two major classes, the **DataFrame** class with two-dimensional data objects and tabular data organized in columns and the class **Series** with a focus on one-dimensional data objects. Both classes allow you to index data easily as we will see in the examples below. \n", + "**pandas** allows you also to perform mathematical operations on the data, spanning from simple reshapings of vectors and matrices to statistical operations. \n", + "\n", + "The following simple example shows how we can, in an easy way make tables of our data. Here we define a data set which includes names, place of birth and date of birth, and displays the data in an easy to read way. We will see repeated use of **pandas**, in particular in connection with classification of data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import pandas as pd\n", + "from IPython.display import display\n", + "data = {'First Name': [\"Frodo\", \"Bilbo\", \"Aragorn II\", \"Samwise\"],\n", + " 'Last Name': [\"Baggins\", \"Baggins\",\"Elessar\",\"Gamgee\"],\n", + " 'Place of birth': [\"Shire\", \"Shire\", \"Eriador\", \"Shire\"],\n", + " 'Date of Birth T.A.': [2968, 2890, 2931, 2980]\n", + " }\n", + "data_pandas = pd.DataFrame(data)\n", + "display(data_pandas)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the above we have imported **pandas** with the shorthand **pd**, the latter has become the standard way we import **pandas**. We make then a list of various variables\n", + "and reorganize the aboves lists into a **DataFrame** and then print out a neat table with specific column labels as *Name*, *place of birth* and *date of birth*.\n", + "Displaying these results, we see that the indices are given by the default numbers from zero to three.\n", + "**pandas** is extremely flexible and we can easily change the above indices by defining a new type of indexing as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam'])\n", + "display(data_pandas)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Thereafter we display the content of the row which begins with the index **Aragorn**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "display(data_pandas.loc['Aragorn'])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can easily append data to this, for example" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "new_hobbit = {'First Name': [\"Peregrin\"],\n", + " 'Last Name': [\"Took\"],\n", + " 'Place of birth': [\"Shire\"],\n", + " 'Date of Birth T.A.': [2990]\n", + " }\n", + "data_pandas=data_pandas.append(pd.DataFrame(new_hobbit, index=['Pippin']))\n", + "display(data_pandas)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here are other examples where we use the **DataFrame** functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix \n", + "of dimensionality $10\\times 5$ and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import pandas as pd\n", + "from IPython.display import display\n", + "np.random.seed(100)\n", + "# setting up a 10 x 5 matrix\n", + "rows = 10\n", + "cols = 5\n", + "a = np.random.randn(rows,cols)\n", + "df = pd.DataFrame(a)\n", + "display(df)\n", + "print(df.mean())\n", + "print(df.std())\n", + "display(df**2)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Thereafter we can select specific columns only and plot final results" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth']\n", + "df.index = np.arange(10)\n", + "\n", + "display(df)\n", + "print(df['Second'].mean() )\n", + "print(df.info())\n", + "print(df.describe())\n", + "\n", + "from pylab import plt, mpl\n", + "plt.style.use('seaborn')\n", + "mpl.rcParams['font.family'] = 'serif'\n", + "\n", + "df.cumsum().plot(lw=2.0, figsize=(10,6))\n", + "plt.show()\n", + "\n", + "\n", + "df.plot.bar(figsize=(10,6), rot=15)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can produce a $4\\times 4$ matrix" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "b = np.arange(16).reshape((4,4))\n", + "print(b)\n", + "df1 = pd.DataFrame(b)\n", + "print(df1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and many other operations. \n", + "\n", + "The **Series** class is another important class included in\n", + "**pandas**. You can view it as a specialization of **DataFrame** but where\n", + "we have just a single column of data. It shares many of the same features as _DataFrame. As with **DataFrame**,\n", + "most operations are vectorized, achieving thereby a high performance when dealing with computations of arrays, in particular labeled arrays.\n", + "As we will see below it leads also to a very concice code close to the mathematical operations we may be interested in.\n", + "For multidimensional arrays, we recommend strongly [xarray](http://xarray.pydata.org/en/stable/). **xarray** has much of the same flexibility as **pandas**, but allows for the extension to higher dimensions than two. We will see examples later of the usage of both **pandas** and **xarray**. \n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "In order to study various Machine Learning algorithms, we need to\n", + "access data. Acccessing data is an essential step in all machine\n", + "learning algorithms. In particular, setting up the so-called **design\n", + "matrix** (to be defined below) is often the first element we need in\n", + "order to perform our calculations. To set up the design matrix means\n", + "reading (and later, when the calculations are done, writing) data\n", + "in various formats, The formats span from reading files from disk,\n", + "loading data from databases and interacting with online sources\n", + "like web application programming interfaces (APIs).\n", + "\n", + "In handling various input formats, as discussed above, we will mainly stay with **pandas**,\n", + "a Python package which allows us, in a seamless and painless way, to\n", + "deal with a multitude of formats, from standard **csv** (comma separated\n", + "values) files, via **excel**, **html** to **hdf5** formats. With **pandas**\n", + "and the **DataFrame** and **Series** functionalities we are able to convert text data\n", + "into the calculational formats we need for a specific algorithm. And our code is going to be \n", + "pretty close the basic mathematical expressions.\n", + "\n", + "Our first data set is going to be a classic from nuclear physics, namely all\n", + "available data on binding energies. Don't be intimidated if you are not familiar with nuclear physics. It serves simply as an example here of a data set. \n", + "\n", + "We will show some of the\n", + "strengths of packages like **Scikit-Learn** in fitting nuclear binding energies to\n", + "specific functions using linear regression first. Then, as a teaser, we will show you how \n", + "you can easily implement other algorithms like decision trees and random forests and neural networks.\n", + "\n", + "But before we really start with nuclear physics data, let's just look at some simpler polynomial fitting cases, such as,\n", + "(don't be offended) fitting straight lines!\n", + "\n", + "\n", + "\n", + "\n", + "## Simple linear regression model using **scikit-learn**\n", + "\n", + "We start with perhaps our simplest possible example, using **Scikit-Learn** to perform linear regression analysis on a data set produced by us. \n", + "\n", + "What follows is a simple Python code where we have defined a function\n", + "$y$ in terms of the variable $x$. Both are defined as vectors with $100$ entries. \n", + "The numbers in the vector $\\hat{x}$ are given\n", + "by random numbers generated with a uniform distribution with entries\n", + "$x_i \\in [0,1]$ (more about probability distribution functions\n", + "later). These values are then used to define a function $y(x)$\n", + "(tabulated again as a vector) with a linear dependence on $x$ plus a\n", + "random noise added via the normal distribution.\n", + "\n", + "\n", + "The Numpy functions are imported used the **import numpy as np**\n", + "statement and the random number generator for the uniform distribution\n", + "is called using the function **np.random.rand()**, where we specificy\n", + "that we want $100$ random variables. Using Numpy we define\n", + "automatically an array with the specified number of elements, $100$ in\n", + "our case. With the Numpy function **randn()** we can compute random\n", + "numbers with the normal distribution (mean value $\\mu$ equal to zero and\n", + "variance $\\sigma^2$ set to one) and produce the values of $y$ assuming a linear\n", + "dependence as function of $x$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y = 2x+N(0,1),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $N(0,1)$ represents random numbers generated by the normal\n", + "distribution. From **Scikit-Learn** we import then the\n", + "**LinearRegression** functionality and make a prediction $\\tilde{y} =\n", + "\\alpha + \\beta x$ using the function **fit(x,y)**. We call the set of\n", + "data $(\\hat{x},\\hat{y})$ for our training data. The Python package\n", + "**scikit-learn** has also a functionality which extracts the above\n", + "fitting parameters $\\alpha$ and $\\beta$ (see below). Later we will\n", + "distinguish between training data and test data.\n", + "\n", + "For plotting we use the Python package\n", + "[matplotlib](https://matplotlib.org/) which produces publication\n", + "quality figures. Feel free to explore the extensive\n", + "[gallery](https://matplotlib.org/gallery/index.html) of examples. In\n", + "this example we plot our original values of $x$ and $y$ as well as the\n", + "prediction **ypredict** ($\\tilde{y}$), which attempts at fitting our\n", + "data with a straight line.\n", + "\n", + "The Python code follows here." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression\n", + "\n", + "x = np.random.rand(100,1)\n", + "y = 2*x+np.random.randn(100,1)\n", + "linreg = LinearRegression()\n", + "linreg.fit(x,y)\n", + "xnew = np.array([[0],[1]])\n", + "ypredict = linreg.predict(xnew)\n", + "\n", + "plt.plot(xnew, ypredict, \"r-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,1.0,0, 5.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Simple Linear Regression')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This example serves several aims. It allows us to demonstrate several\n", + "aspects of data analysis and later machine learning algorithms. The\n", + "immediate visualization shows that our linear fit is not\n", + "impressive. It goes through the data points, but there are many\n", + "outliers which are not reproduced by our linear regression. We could\n", + "now play around with this small program and change for example the\n", + "factor in front of $x$ and the normal distribution. Try to change the\n", + "function $y$ to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y = 10x+0.01 \\times N(0,1),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $x$ is defined as before. Does the fit look better? Indeed, by\n", + "reducing the role of the noise given by the normal distribution we see immediately that\n", + "our linear prediction seemingly reproduces better the training\n", + "set. However, this testing 'by the eye' is obviouly not satisfactory in the\n", + "long run. Here we have only defined the training data and our model, and \n", + "have not discussed a more rigorous approach to the **cost** function.\n", + "\n", + "We need more rigorous criteria in defining whether we have succeeded or\n", + "not in modeling our training data. You will be surprised to see that\n", + "many scientists seldomly venture beyond this 'by the eye' approach. A\n", + "standard approach for the *cost* function is the so-called $\\chi^2$\n", + "function (a variant of the mean-squared error (MSE))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\chi^2 = \\frac{1}{n}\n", + "\\sum_{i=0}^{n-1}\\frac{(y_i-\\tilde{y}_i)^2}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\sigma_i^2$ is the variance (to be defined later) of the entry\n", + "$y_i$. We may not know the explicit value of $\\sigma_i^2$, it serves\n", + "however the aim of scaling the equations and make the cost function\n", + "dimensionless. \n", + "\n", + "Minimizing the cost function is a central aspect of\n", + "our discussions to come. Finding its minima as function of the model\n", + "parameters ($\\alpha$ and $\\beta$ in our case) will be a recurring\n", + "theme in these series of lectures. Essentially all machine learning\n", + "algorithms we will discuss center around the minimization of the\n", + "chosen cost function. This depends in turn on our specific\n", + "model for describing the data, a typical situation in supervised\n", + "learning. Automatizing the search for the minima of the cost function is a\n", + "central ingredient in all algorithms. Typical methods which are\n", + "employed are various variants of **gradient** methods. These will be\n", + "discussed in more detail later. Again, you'll be surprised to hear that\n", + "many practitioners minimize the above function ''by the eye', popularly dubbed as \n", + "'chi by the eye'. That is, change a parameter and see (visually and numerically) that \n", + "the $\\chi^2$ function becomes smaller. \n", + "\n", + "There are many ways to define the cost function. A simpler approach is to look at the relative difference between the training data and the predicted data, that is we define \n", + "the relative error (why would we prefer the MSE instead of the relative error?) as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\epsilon_{\\mathrm{relative}}= \\frac{\\vert \\hat{y} -\\hat{\\tilde{y}}\\vert}{\\vert \\hat{y}\\vert}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The squared cost function results in an arithmetic mean-unbiased\n", + "estimator, and the absolute-value cost function results in a\n", + "median-unbiased estimator (in the one-dimensional case, and a\n", + "geometric median-unbiased estimator for the multi-dimensional\n", + "case). The squared cost function has the disadvantage that it has the tendency\n", + "to be dominated by outliers.\n", + "\n", + "We can modify easily the above Python code and plot the relative error instead" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression\n", + "\n", + "x = np.random.rand(100,1)\n", + "y = 5*x+0.01*np.random.randn(100,1)\n", + "linreg = LinearRegression()\n", + "linreg.fit(x,y)\n", + "ypredict = linreg.predict(x)\n", + "\n", + "plt.plot(x, np.abs(ypredict-y)/abs(y), \"ro\")\n", + "plt.axis([0,1.0,0.0, 0.5])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$\\epsilon_{\\mathrm{relative}}$')\n", + "plt.title(r'Relative error')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Depending on the parameter in front of the normal distribution, we may\n", + "have a small or larger relative error. Try to play around with\n", + "different training data sets and study (graphically) the value of the\n", + "relative error.\n", + "\n", + "As mentioned above, **Scikit-Learn** has an impressive functionality.\n", + "We can for example extract the values of $\\alpha$ and $\\beta$ and\n", + "their error estimates, or the variance and standard deviation and many\n", + "other properties from the statistical data analysis. \n", + "\n", + "Here we show an\n", + "example of the functionality of **Scikit-Learn**." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np \n", + "import matplotlib.pyplot as plt \n", + "from sklearn.linear_model import LinearRegression \n", + "from sklearn.metrics import mean_squared_error, r2_score, mean_squared_log_error, mean_absolute_error\n", + "\n", + "x = np.random.rand(100,1)\n", + "y = 2.0+ 5*x+0.5*np.random.randn(100,1)\n", + "linreg = LinearRegression()\n", + "linreg.fit(x,y)\n", + "ypredict = linreg.predict(x)\n", + "print('The intercept alpha: \\n', linreg.intercept_)\n", + "print('Coefficient beta : \\n', linreg.coef_)\n", + "# The mean squared error \n", + "print(\"Mean squared error: %.2f\" % mean_squared_error(y, ypredict))\n", + "# Explained variance score: 1 is perfect prediction \n", + "print('Variance score: %.2f' % r2_score(y, ypredict))\n", + "# Mean squared log error \n", + "print('Mean squared log error: %.2f' % mean_squared_log_error(y, ypredict) )\n", + "# Mean absolute error \n", + "print('Mean absolute error: %.2f' % mean_absolute_error(y, ypredict))\n", + "plt.plot(x, ypredict, \"r-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0.0,1.0,1.5, 7.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Linear Regression fit ')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The function **coef** gives us the parameter $\\beta$ of our fit while **intercept** yields \n", + "$\\alpha$. Depending on the constant in front of the normal distribution, we get values near or far from $alpha =2$ and $\\beta =5$. Try to play around with different parameters in front of the normal distribution. The function **meansquarederror** gives us the mean square error, a risk metric corresponding to the expected value of the squared (quadratic) error or loss defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "MSE(\\hat{y},\\hat{\\tilde{y}}) = \\frac{1}{n}\n", + "\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The smaller the value, the better the fit. Ideally we would like to\n", + "have an MSE equal zero. The attentive reader has probably recognized\n", + "this function as being similar to the $\\chi^2$ function defined above.\n", + "\n", + "The **r2score** function computes $R^2$, the coefficient of\n", + "determination. It provides a measure of how well future samples are\n", + "likely to be predicted by the model. Best possible score is 1.0 and it\n", + "can be negative (because the model can be arbitrarily worse). A\n", + "constant model that always predicts the expected value of $\\hat{y}$,\n", + "disregarding the input features, would get a $R^2$ score of $0.0$.\n", + "\n", + "If $\\tilde{\\hat{y}}_i$ is the predicted value of the $i-th$ sample and $y_i$ is the corresponding true value, then the score $R^2$ is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "R^2(\\hat{y}, \\tilde{\\hat{y}}) = 1 - \\frac{\\sum_{i=0}^{n - 1} (y_i - \\tilde{y}_i)^2}{\\sum_{i=0}^{n - 1} (y_i - \\bar{y})^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have defined the mean value of $\\hat{y}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\bar{y} = \\frac{1}{n} \\sum_{i=0}^{n - 1} y_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Another quantity taht we will meet again in our discussions of regression analysis is \n", + " the mean absolute error (MAE), a risk metric corresponding to the expected value of the absolute error loss or what we call the $l1$-norm loss. In our discussion above we presented the relative error.\n", + "The MAE is defined as follows" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\text{MAE}(\\hat{y}, \\hat{\\tilde{y}}) = \\frac{1}{n} \\sum_{i=0}^{n-1} \\left| y_i - \\tilde{y}_i \\right|.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We present the \n", + "squared logarithmic (quadratic) error" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\text{MSLE}(\\hat{y}, \\hat{\\tilde{y}}) = \\frac{1}{n} \\sum_{i=0}^{n - 1} (\\log_e (1 + y_i) - \\log_e (1 + \\tilde{y}_i) )^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\log_e (x)$ stands for the natural logarithm of $x$. This error\n", + "estimate is best to use when targets having exponential growth, such\n", + "as population counts, average sales of a commodity over a span of\n", + "years etc. \n", + "\n", + "\n", + "Finally, another cost function is the Huber cost function used in robust regression.\n", + "\n", + "The rationale behind this possible cost function is its reduced\n", + "sensitivity to outliers in the data set. In our discussions on\n", + "dimensionality reduction and normalization of data we will meet other\n", + "ways of dealing with outliers.\n", + "\n", + "The Huber cost function is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "H_{\\delta}(a)={\\begin{cases}{\\frac {1}{2}}{a^{2}}&{\\text{for }}|a|\\leq \\delta ,\\\\\\delta (|a|-{\\frac {1}{2}}\\delta ),&{\\text{otherwise.}}\\end{cases}}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here $a=\\boldsymbol{y} - \\boldsymbol{\\tilde{y}}$.\n", + "We will discuss in more\n", + "detail these and other functions in the various lectures. We conclude this part with another example. Instead of \n", + "a linear $x$-dependence we study now a cubic polynomial and use the polynomial regression analysis tools of scikit-learn." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "import random\n", + "from sklearn.linear_model import Ridge\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "from sklearn.pipeline import make_pipeline\n", + "from sklearn.linear_model import LinearRegression\n", + "\n", + "x=np.linspace(0.02,0.98,200)\n", + "noise = np.asarray(random.sample((range(200)),200))\n", + "y=x**3*noise\n", + "yn=x**3*100\n", + "poly3 = PolynomialFeatures(degree=3)\n", + "X = poly3.fit_transform(x[:,np.newaxis])\n", + "clf3 = LinearRegression()\n", + "clf3.fit(X,y)\n", + "\n", + "Xplot=poly3.fit_transform(x[:,np.newaxis])\n", + "poly3_plot=plt.plot(x, clf3.predict(Xplot), label='Cubic Fit')\n", + "plt.plot(x,yn, color='red', label=\"True Cubic\")\n", + "plt.scatter(x, y, label='Data', color='orange', s=15)\n", + "plt.legend()\n", + "plt.show()\n", + "\n", + "def error(a):\n", + " for i in y:\n", + " err=(y-yn)/yn\n", + " return abs(np.sum(err))/len(err)\n", + "\n", + "print (error(y))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Let us now dive into nuclear physics and remind ourselves briefly about some basic features about binding\n", + "energies. A basic quantity which can be measured for the ground\n", + "states of nuclei is the atomic mass $M(N, Z)$ of the neutral atom with\n", + "atomic mass number $A$ and charge $Z$. The number of neutrons is $N$. There are indeed several sophisticated experiments worldwide which allow us to measure this quantity to high precision (parts per million even). \n", + "\n", + "Atomic masses are usually tabulated in terms of the mass excess defined by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\Delta M(N, Z) = M(N, Z) - uA,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $u$ is the Atomic Mass Unit" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "u = M(^{12}\\mathrm{C})/12 = 931.4940954(57) \\hspace{0.1cm} \\mathrm{MeV}/c^2.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The nucleon masses are" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "m_p = 1.00727646693(9)u,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "m_n = 939.56536(8)\\hspace{0.1cm} \\mathrm{MeV}/c^2 = 1.0086649156(6)u.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the [2016 mass evaluation of by W.J.Huang, G.Audi, M.Wang, F.G.Kondev, S.Naimi and X.Xu](http://nuclearmasses.org/resources_folder/Wang_2017_Chinese_Phys_C_41_030003.pdf)\n", + "there are data on masses and decays of 3437 nuclei.\n", + "\n", + "The nuclear binding energy is defined as the energy required to break\n", + "up a given nucleus into its constituent parts of $N$ neutrons and $Z$\n", + "protons. In terms of the atomic masses $M(N, Z)$ the binding energy is\n", + "defined by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "BE(N, Z) = ZM_H c^2 + Nm_n c^2 - M(N, Z)c^2 ,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $M_H$ is the mass of the hydrogen atom and $m_n$ is the mass of the neutron.\n", + "In terms of the mass excess the binding energy is given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "BE(N, Z) = Z\\Delta_H c^2 + N\\Delta_n c^2 -\\Delta(N, Z)c^2 ,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\Delta_H c^2 = 7.2890$ MeV and $\\Delta_n c^2 = 8.0713$ MeV.\n", + "\n", + "\n", + "A popular and physically intuitive model which can be used to parametrize \n", + "the experimental binding energies as function of $A$, is the so-called \n", + "**liquid drop model**. The ansatz is based on the following expression" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "BE(N,Z) = a_1A-a_2A^{2/3}-a_3\\frac{Z^2}{A^{1/3}}-a_4\\frac{(N-Z)^2}{A},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $A$ stands for the number of nucleons and the $a_i$s are parameters which are determined by a fit \n", + "to the experimental data. \n", + "\n", + "\n", + "\n", + "\n", + "To arrive at the above expression we have assumed that we can make the following assumptions:\n", + "\n", + " * There is a volume term $a_1A$ proportional with the number of nucleons (the energy is also an extensive quantity). When an assembly of nucleons of the same size is packed together into the smallest volume, each interior nucleon has a certain number of other nucleons in contact with it. This contribution is proportional to the volume.\n", + "\n", + " * There is a surface energy term $a_2A^{2/3}$. The assumption here is that a nucleon at the surface of a nucleus interacts with fewer other nucleons than one in the interior of the nucleus and hence its binding energy is less. This surface energy term takes that into account and is therefore negative and is proportional to the surface area.\n", + "\n", + " * There is a Coulomb energy term $a_3\\frac{Z^2}{A^{1/3}}$. The electric repulsion between each pair of protons in a nucleus yields less binding. \n", + "\n", + " * There is an asymmetry term $a_4\\frac{(N-Z)^2}{A}$. This term is associated with the Pauli exclusion principle and reflects the fact that the proton-neutron interaction is more attractive on the average than the neutron-neutron and proton-proton interactions.\n", + "\n", + "We could also add a so-called pairing term, which is a correction term that\n", + "arises from the tendency of proton pairs and neutron pairs to\n", + "occur. An even number of particles is more stable than an odd number. \n", + "\n", + "\n", + "### Organizing our data\n", + "\n", + "Let us start with reading and organizing our data. \n", + "We start with the compilation of masses and binding energies from 2016.\n", + "After having downloaded this file to our own computer, we are now ready to read the file and start structuring our data.\n", + "\n", + "\n", + "We start with preparing folders for storing our calculations and the data file over masses and binding energies. We import also various modules that we will find useful in order to present various Machine Learning methods. Here we focus mainly on the functionality of **scikit-learn**." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "import sklearn.linear_model as skl\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\n", + "import os\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"MassEval2016.dat\"),'r')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Before we proceed, we define also a function for making our plots. You can obviously avoid this and simply set up various **matplotlib** commands every time you need them. You may however find it convenient to collect all such commands in one function and simply call this function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from pylab import plt, mpl\n", + "plt.style.use('seaborn')\n", + "mpl.rcParams['font.family'] = 'serif'\n", + "\n", + "def MakePlot(x,y, styles, labels, axlabels):\n", + " plt.figure(figsize=(10,6))\n", + " for i in range(len(x)):\n", + " plt.plot(x[i], y[i], styles[i], label = labels[i])\n", + " plt.xlabel(axlabels[0])\n", + " plt.ylabel(axlabels[1])\n", + " plt.legend(loc=0)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Our next step is to read the data on experimental binding energies and\n", + "reorganize them as functions of the mass number $A$, the number of\n", + "protons $Z$ and neutrons $N$ using **pandas**. Before we do this it is\n", + "always useful (unless you have a binary file or other types of compressed\n", + "data) to actually open the file and simply take a look at it!\n", + "\n", + "\n", + "In particular, the program that outputs the final nuclear masses is written in Fortran with a specific format. It means that we need to figure out the format and which columns contain the data we are interested in. Pandas comes with a function that reads formatted output. After having admired the file, we are now ready to start massaging it with **pandas**. The file begins with some basic format information." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\"\"\" \n", + "This is taken from the data file of the mass 2016 evaluation. \n", + "All files are 3436 lines long with 124 character per line. \n", + " Headers are 39 lines long. \n", + " col 1 : Fortran character control: 1 = page feed 0 = line feed \n", + " format : a1,i3,i5,i5,i5,1x,a3,a4,1x,f13.5,f11.5,f11.3,f9.3,1x,a2,f11.3,f9.3,1x,i3,1x,f12.5,f11.5 \n", + " These formats are reflected in the pandas widths variable below, see the statement \n", + " widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1), \n", + " Pandas has also a variable header, with length 39 in this case. \n", + "\"\"\"" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The data we are interested in are in columns 2, 3, 4 and 11, giving us\n", + "the number of neutrons, protons, mass numbers and binding energies,\n", + "respectively. We add also for the sake of completeness the element name. The data are in fixed-width formatted lines and we will\n", + "covert them into the **pandas** DataFrame structure." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Read the experimental data with Pandas\n", + "Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11),\n", + " names=('N', 'Z', 'A', 'Element', 'Ebinding'),\n", + " widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),\n", + " header=39,\n", + " index_col=False)\n", + "\n", + "# Extrapolated values are indicated by '#' in place of the decimal place, so\n", + "# the Ebinding column won't be numeric. Coerce to float and drop these entries.\n", + "Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce')\n", + "Masses = Masses.dropna()\n", + "# Convert from keV to MeV.\n", + "Masses['Ebinding'] /= 1000\n", + "\n", + "# Group the DataFrame by nucleon number, A.\n", + "Masses = Masses.groupby('A')\n", + "# Find the rows of the grouped DataFrame with the maximum binding energy.\n", + "Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We have now read in the data, grouped them according to the variables we are interested in. \n", + "We see how easy it is to reorganize the data using **pandas**. If we\n", + "were to do these operations in C/C++ or Fortran, we would have had to\n", + "write various functions/subroutines which perform the above\n", + "reorganizations for us. Having reorganized the data, we can now start\n", + "to make some simple fits using both the functionalities in **numpy** and\n", + "**Scikit-Learn** afterwards. \n", + "\n", + "Now we define five variables which contain\n", + "the number of nucleons $A$, the number of protons $Z$ and the number of neutrons $N$, the element name and finally the energies themselves." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "A = Masses['A']\n", + "Z = Masses['Z']\n", + "N = Masses['N']\n", + "Element = Masses['Element']\n", + "Energies = Masses['Ebinding']\n", + "print(Masses)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The next step, and we will define this mathematically later, is to set up the so-called **design matrix**. We will throughout call this matrix $\\boldsymbol{X}$.\n", + "It has dimensionality $p\\times n$, where $n$ is the number of data points and $p$ are the so-called predictors. In our case here they are given by the number of polynomials in $A$ we wish to include in the fit." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Now we set up the design matrix X\n", + "X = np.zeros((len(A),5))\n", + "X[:,0] = 1\n", + "X[:,1] = A\n", + "X[:,2] = A**(2.0/3.0)\n", + "X[:,3] = A**(-1.0/3.0)\n", + "X[:,4] = A**(-1.0)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With **scikitlearn** we are now ready to use linear regression and fit our data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "clf = skl.LinearRegression().fit(X, Energies)\n", + "fity = clf.predict(X)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Pretty simple! \n", + "Now we can print measures of how our fit is doing, the coefficients from the fits and plot the final fit together with our data." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# The mean squared error \n", + "print(\"Mean squared error: %.2f\" % mean_squared_error(Energies, fity))\n", + "# Explained variance score: 1 is perfect prediction \n", + "print('Variance score: %.2f' % r2_score(Energies, fity))\n", + "# Mean absolute error \n", + "print('Mean absolute error: %.2f' % mean_absolute_error(Energies, fity))\n", + "print(clf.coef_, clf.intercept_)\n", + "\n", + "Masses['Eapprox'] = fity\n", + "# Generate a plot comparing the experimental with the fitted values values.\n", + "fig, ax = plt.subplots()\n", + "ax.set_xlabel(r'$A = N + Z$')\n", + "ax.set_ylabel(r'$E_\\mathrm{bind}\\,/\\mathrm{MeV}$')\n", + "ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,\n", + " label='Ame2016')\n", + "ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',\n", + " label='Fit')\n", + "ax.legend()\n", + "save_fig(\"Masses2016\")\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "As a teaser, let us now see how we can do this with decision trees using **scikit-learn**. Later we will switch to so-called **random forests**!" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\n", + "#Decision Tree Regression\n", + "from sklearn.tree import DecisionTreeRegressor\n", + "regr_1=DecisionTreeRegressor(max_depth=5)\n", + "regr_2=DecisionTreeRegressor(max_depth=7)\n", + "regr_3=DecisionTreeRegressor(max_depth=9)\n", + "regr_1.fit(X, Energies)\n", + "regr_2.fit(X, Energies)\n", + "regr_3.fit(X, Energies)\n", + "\n", + "\n", + "y_1 = regr_1.predict(X)\n", + "y_2 = regr_2.predict(X)\n", + "y_3=regr_3.predict(X)\n", + "Masses['Eapprox'] = y_3\n", + "# Plot the results\n", + "plt.figure()\n", + "plt.plot(A, Energies, color=\"blue\", label=\"Data\", linewidth=2)\n", + "plt.plot(A, y_1, color=\"red\", label=\"max_depth=5\", linewidth=2)\n", + "plt.plot(A, y_2, color=\"green\", label=\"max_depth=7\", linewidth=2)\n", + "plt.plot(A, y_3, color=\"m\", label=\"max_depth=9\", linewidth=2)\n", + "\n", + "plt.xlabel(\"$A$\")\n", + "plt.ylabel(\"$E$[MeV]\")\n", + "plt.title(\"Decision Tree Regression\")\n", + "plt.legend()\n", + "save_fig(\"Masses2016Trees\")\n", + "plt.show()\n", + "print(Masses)\n", + "print(np.mean( (Energies-y_1)**2))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The **seaborn** package allows us to visualize data in an efficient way. Note that we use **scikit-learn**'s multi-layer perceptron (or feed forward neural network) \n", + "functionality." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from sklearn.neural_network import MLPRegressor\n", + "from sklearn.metrics import accuracy_score\n", + "import seaborn as sns\n", + "\n", + "X_train = X\n", + "Y_train = Energies\n", + "n_hidden_neurons = 100\n", + "epochs = 100\n", + "# store models for later use\n", + "eta_vals = np.logspace(-5, 1, 7)\n", + "lmbd_vals = np.logspace(-5, 1, 7)\n", + "# store the models for later use\n", + "DNN_scikit = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object)\n", + "train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals)))\n", + "sns.set()\n", + "for i, eta in enumerate(eta_vals):\n", + " for j, lmbd in enumerate(lmbd_vals):\n", + " dnn = MLPRegressor(hidden_layer_sizes=(n_hidden_neurons), activation='logistic',\n", + " alpha=lmbd, learning_rate_init=eta, max_iter=epochs)\n", + " dnn.fit(X_train, Y_train)\n", + " DNN_scikit[i][j] = dnn\n", + " train_accuracy[i][j] = dnn.score(X_train, Y_train)\n", + "\n", + "fig, ax = plt.subplots(figsize = (10, 10))\n", + "sns.heatmap(train_accuracy, annot=True, ax=ax, cmap=\"viridis\")\n", + "ax.set_title(\"Training Accuracy\")\n", + "ax.set_ylabel(\"$\\eta$\")\n", + "ax.set_xlabel(\"$\\lambda$\")\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Linear Regression, basic elements\n", + "\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureAug27.mp4?vrtx=view-as-webpage).\n", + "\n", + "\n", + "Fitting a continuous function with linear parameterization in terms of the parameters $\\boldsymbol{\\beta}$.\n", + "* Method of choice for fitting a continuous function!\n", + "\n", + "* Gives an excellent introduction to central Machine Learning features with **understandable pedagogical** links to other methods like **Neural Networks**, **Support Vector Machines** etc\n", + "\n", + "* Analytical expression for the fitting parameters $\\boldsymbol{\\beta}$\n", + "\n", + "* Analytical expressions for statistical propertiers like mean values, variances, confidence intervals and more\n", + "\n", + "* Analytical relation with probabilistic interpretations \n", + "\n", + "* Easy to introduce basic concepts like bias-variance tradeoff, cross-validation, resampling and regularization techniques and many other ML topics\n", + "\n", + "* Easy to code! And links well with classification problems and logistic regression and neural networks\n", + "\n", + "* Allows for **easy** hands-on understanding of gradient descent methods\n", + "\n", + "* and many more features\n", + "\n", + "For more discussions of Ridge and Lasso regression, [Wessel van Wieringen's](https://arxiv.org/abs/1509.09169) article is highly recommended.\n", + "Similarly, [Mehta et al's article](https://arxiv.org/abs/1803.08823) is also recommended.\n", + "\n", + "\n", + "\n", + "Regression modeling deals with the description of the sampling distribution of a given random variable $y$ and how it varies as function of another variable or a set of such variables $\\boldsymbol{x} =[x_0, x_1,\\dots, x_{n-1}]^T$. \n", + "The first variable is called the **dependent**, the **outcome** or the **response** variable while the set of variables $\\boldsymbol{x}$ is called the independent variable, or the predictor variable or the explanatory variable. \n", + "\n", + "A regression model aims at finding a likelihood function $p(\\boldsymbol{y}\\vert \\boldsymbol{x})$, that is the conditional distribution for $\\boldsymbol{y}$ with a given $\\boldsymbol{x}$. The estimation of $p(\\boldsymbol{y}\\vert \\boldsymbol{x})$ is made using a data set with \n", + "* $n$ cases $i = 0, 1, 2, \\dots, n-1$ \n", + "\n", + "* Response (target, dependent or outcome) variable $y_i$ with $i = 0, 1, 2, \\dots, n-1$ \n", + "\n", + "* $p$ so-called explanatory (independent or predictor) variables $\\boldsymbol{x}_i=[x_{i0}, x_{i1}, \\dots, x_{ip-1}]$ with $i = 0, 1, 2, \\dots, n-1$ and explanatory variables running from $0$ to $p-1$. See below for more explicit examples. \n", + "\n", + " The goal of the regression analysis is to extract/exploit relationship between $\\boldsymbol{y}$ and $\\boldsymbol{x}$ in or to infer causal dependencies, approximations to the likelihood functions, functional relationships and to make predictions, making fits and many other things.\n", + "\n", + "\n", + "Consider an experiment in which $p$ characteristics of $n$ samples are\n", + "measured. The data from this experiment, for various explanatory variables $p$ are normally represented by a matrix \n", + "$\\mathbf{X}$.\n", + "\n", + "The matrix $\\mathbf{X}$ is called the *design\n", + "matrix*. Additional information of the samples is available in the\n", + "form of $\\boldsymbol{y}$ (also as above). The variable $\\boldsymbol{y}$ is\n", + "generally referred to as the *response variable*. The aim of\n", + "regression analysis is to explain $\\boldsymbol{y}$ in terms of\n", + "$\\boldsymbol{X}$ through a functional relationship like $y_i =\n", + "f(\\mathbf{X}_{i,\\ast})$. When no prior knowledge on the form of\n", + "$f(\\cdot)$ is available, it is common to assume a linear relationship\n", + "between $\\boldsymbol{X}$ and $\\boldsymbol{y}$. This assumption gives rise to\n", + "the *linear regression model* where $\\boldsymbol{\\beta} = [\\beta_0, \\ldots,\n", + "\\beta_{p-1}]^{T}$ are the *regression parameters*. \n", + "\n", + "Linear regression gives us a set of analytical equations for the parameters $\\beta_j$.\n", + "\n", + "\n", + "In order to understand the relation among the predictors $p$, the set of data $n$ and the target (outcome, output etc) $\\boldsymbol{y}$,\n", + "consider the model we discussed for describing nuclear binding energies. \n", + "\n", + "There we assumed that we could parametrize the data using a polynomial approximation based on the liquid drop model.\n", + "Assuming" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "BE(A) = a_0+a_1A+a_2A^{2/3}+a_3A^{-1/3}+a_4A^{-1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we have five predictors, that is the intercept, the $A$ dependent term, the $A^{2/3}$ term and the $A^{-1/3}$ and $A^{-1}$ terms.\n", + "This gives $p=0,1,2,3,4$. Furthermore we have $n$ entries for each predictor. It means that our design matrix is a \n", + "$p\\times n$ matrix $\\boldsymbol{X}$.\n", + "\n", + "Here the predictors are based on a model we have made. A popular data set which is widely encountered in ML applications is the\n", + "so-called [credit card default data from Taiwan](https://www.sciencedirect.com/science/article/pii/S0957417407006719?via%3Dihub). The data set contains data on $n=30000$ credit card holders with predictors like gender, marital status, age, profession, education, etc. In total there are $24$ such predictors or attributes leading to a design matrix of dimensionality $24 \\times 30000$. This is however a classification problem and we will come back to it when we discuss Logistic Regression. \n", + "\n", + "\n", + "Before we proceed let us study a case from linear algebra where we aim at fitting a set of data $\\boldsymbol{y}=[y_0,y_1,\\dots,y_{n-1}]$. We could think of these data as a result of an experiment or a complicated numerical experiment. These data are functions of a series of variables $\\boldsymbol{x}=[x_0,x_1,\\dots,x_{n-1}]$, that is $y_i = y(x_i)$ with $i=0,1,2,\\dots,n-1$. The variables $x_i$ could represent physical quantities like time, temperature, position etc. We assume that $y(x)$ is a smooth function. \n", + "\n", + "Since obtaining these data points may not be trivial, we want to use these data to fit a function which can allow us to make predictions for values of $y$ which are not in the present set. The perhaps simplest approach is to assume we can parametrize our function in terms of a polynomial of degree $n-1$ with $n$ points, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y=y(x) \\rightarrow y(x_i)=\\tilde{y}_i+\\epsilon_i=\\sum_{j=0}^{n-1} \\beta_j x_i^j+\\epsilon_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\epsilon_i$ is the error in our approximation. \n", + "\n", + "\n", + "For every set of values $y_i,x_i$ we have thus the corresponding set of equations" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "y_0&=\\beta_0+\\beta_1x_0^1+\\beta_2x_0^2+\\dots+\\beta_{n-1}x_0^{n-1}+\\epsilon_0\\\\\n", + "y_1&=\\beta_0+\\beta_1x_1^1+\\beta_2x_1^2+\\dots+\\beta_{n-1}x_1^{n-1}+\\epsilon_1\\\\\n", + "y_2&=\\beta_0+\\beta_1x_2^1+\\beta_2x_2^2+\\dots+\\beta_{n-1}x_2^{n-1}+\\epsilon_2\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{n-1}&=\\beta_0+\\beta_1x_{n-1}^1+\\beta_2x_{n-1}^2+\\dots+\\beta_{n-1}x_{n-1}^{n-1}+\\epsilon_{n-1}.\\\\\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Defining the vectors" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y} = [y_0,y_1, y_2,\\dots, y_{n-1}]^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta} = [\\beta_0,\\beta_1, \\beta_2,\\dots, \\beta_{n-1}]^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\epsilon} = [\\epsilon_0,\\epsilon_1, \\epsilon_2,\\dots, \\epsilon_{n-1}]^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and the design matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\n", + "\\begin{bmatrix} \n", + "1& x_{0}^1 &x_{0}^2& \\dots & \\dots &x_{0}^{n-1}\\\\\n", + "1& x_{1}^1 &x_{1}^2& \\dots & \\dots &x_{1}^{n-1}\\\\\n", + "1& x_{2}^1 &x_{2}^2& \\dots & \\dots &x_{2}^{n-1}\\\\ \n", + "\\dots& \\dots &\\dots& \\dots & \\dots &\\dots\\\\\n", + "1& x_{n-1}^1 &x_{n-1}^2& \\dots & \\dots &x_{n-1}^{n-1}\\\\\n", + "\\end{bmatrix}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we can rewrite our equations as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y} = \\boldsymbol{X}\\boldsymbol{\\beta}+\\boldsymbol{\\epsilon}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The above design matrix is called a [Vandermonde matrix](https://en.wikipedia.org/wiki/Vandermonde_matrix).\n", + "\n", + "We are obviously not limited to the above polynomial expansions. We\n", + "could replace the various powers of $x$ with elements of Fourier\n", + "series or instead of $x_i^j$ we could have $\\cos{(j x_i)}$ or $\\sin{(j\n", + "x_i)}$, or time series or other orthogonal functions. For every set\n", + "of values $y_i,x_i$ we can then generalize the equations to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "y_0&=\\beta_0x_{00}+\\beta_1x_{01}+\\beta_2x_{02}+\\dots+\\beta_{n-1}x_{0n-1}+\\epsilon_0\\\\\n", + "y_1&=\\beta_0x_{10}+\\beta_1x_{11}+\\beta_2x_{12}+\\dots+\\beta_{n-1}x_{1n-1}+\\epsilon_1\\\\\n", + "y_2&=\\beta_0x_{20}+\\beta_1x_{21}+\\beta_2x_{22}+\\dots+\\beta_{n-1}x_{2n-1}+\\epsilon_2\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{i}&=\\beta_0x_{i0}+\\beta_1x_{i1}+\\beta_2x_{i2}+\\dots+\\beta_{n-1}x_{in-1}+\\epsilon_i\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{n-1}&=\\beta_0x_{n-1,0}+\\beta_1x_{n-1,2}+\\beta_2x_{n-1,2}+\\dots+\\beta_{n-1}x_{n-1,n-1}+\\epsilon_{n-1}.\\\\\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "**Note that we have $p=n$ here. The matrix is symmetric. This is generally not the case!**\n", + "\n", + "We redefine in turn the matrix $\\boldsymbol{X}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\n", + "\\begin{bmatrix} \n", + "x_{00}& x_{01} &x_{02}& \\dots & \\dots &x_{0,n-1}\\\\\n", + "x_{10}& x_{11} &x_{12}& \\dots & \\dots &x_{1,n-1}\\\\\n", + "x_{20}& x_{21} &x_{22}& \\dots & \\dots &x_{2,n-1}\\\\ \n", + "\\dots& \\dots &\\dots& \\dots & \\dots &\\dots\\\\\n", + "x_{n-1,0}& x_{n-1,1} &x_{n-1,2}& \\dots & \\dots &x_{n-1,n-1}\\\\\n", + "\\end{bmatrix}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and without loss of generality we rewrite again our equations as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y} = \\boldsymbol{X}\\boldsymbol{\\beta}+\\boldsymbol{\\epsilon}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The left-hand side of this equation is kwown. Our error vector $\\boldsymbol{\\epsilon}$ and the parameter vector $\\boldsymbol{\\beta}$ are our unknow quantities. How can we obtain the optimal set of $\\beta_i$ values? \n", + "\n", + "We have defined the matrix $\\boldsymbol{X}$ via the equations" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "y_0&=\\beta_0x_{00}+\\beta_1x_{01}+\\beta_2x_{02}+\\dots+\\beta_{n-1}x_{0n-1}+\\epsilon_0\\\\\n", + "y_1&=\\beta_0x_{10}+\\beta_1x_{11}+\\beta_2x_{12}+\\dots+\\beta_{n-1}x_{1n-1}+\\epsilon_1\\\\\n", + "y_2&=\\beta_0x_{20}+\\beta_1x_{21}+\\beta_2x_{22}+\\dots+\\beta_{n-1}x_{2n-1}+\\epsilon_1\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{i}&=\\beta_0x_{i0}+\\beta_1x_{i1}+\\beta_2x_{i2}+\\dots+\\beta_{n-1}x_{in-1}+\\epsilon_1\\\\\n", + "\\dots & \\dots \\\\\n", + "y_{n-1}&=\\beta_0x_{n-1,0}+\\beta_1x_{n-1,2}+\\beta_2x_{n-1,2}+\\dots+\\beta_{n-1}x_{n-1,n-1}+\\epsilon_{n-1}.\\\\\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "As we noted above, we stayed with a system with the design matrix \n", + " $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times n}$, that is we have $p=n$. For reasons to come later (algorithmic arguments) we will hereafter define \n", + "our matrix as $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$, with the predictors refering to the column numbers and the entries $n$ being the row elements.\n", + "\n", + "In our [introductory notes](https://compphysics.github.io/MachineLearning/doc/pub/How2ReadData/html/How2ReadData.html) we looked at the so-called [liquid drop model](https://en.wikipedia.org/wiki/Semi-empirical_mass_formula). Let us remind ourselves about what we did by looking at the code.\n", + "\n", + "We restate the parts of the code we are most interested in." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from IPython.display import display\n", + "import os\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"MassEval2016.dat\"),'r')\n", + "\n", + "\n", + "# Read the experimental data with Pandas\n", + "Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11),\n", + " names=('N', 'Z', 'A', 'Element', 'Ebinding'),\n", + " widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1),\n", + " header=39,\n", + " index_col=False)\n", + "\n", + "# Extrapolated values are indicated by '#' in place of the decimal place, so\n", + "# the Ebinding column won't be numeric. Coerce to float and drop these entries.\n", + "Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce')\n", + "Masses = Masses.dropna()\n", + "# Convert from keV to MeV.\n", + "Masses['Ebinding'] /= 1000\n", + "\n", + "# Group the DataFrame by nucleon number, A.\n", + "Masses = Masses.groupby('A')\n", + "# Find the rows of the grouped DataFrame with the maximum binding energy.\n", + "Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()])\n", + "A = Masses['A']\n", + "Z = Masses['Z']\n", + "N = Masses['N']\n", + "Element = Masses['Element']\n", + "Energies = Masses['Ebinding']\n", + "\n", + "# Now we set up the design matrix X\n", + "X = np.zeros((len(A),5))\n", + "X[:,0] = 1\n", + "X[:,1] = A\n", + "X[:,2] = A**(2.0/3.0)\n", + "X[:,3] = A**(-1.0/3.0)\n", + "X[:,4] = A**(-1.0)\n", + "# Then nice printout using pandas\n", + "DesignMatrix = pd.DataFrame(X)\n", + "DesignMatrix.index = A\n", + "DesignMatrix.columns = ['1', 'A', 'A^(2/3)', 'A^(-1/3)', '1/A']\n", + "display(DesignMatrix)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With $\\boldsymbol{\\beta}\\in {\\mathbb{R}}^{p\\times 1}$, it means that we will hereafter write our equations for the approximation as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\tilde{y}}= \\boldsymbol{X}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "throughout these lectures. \n", + "\n", + "With the above we use the design matrix to define the approximation $\\boldsymbol{\\tilde{y}}$ via the unknown quantity $\\boldsymbol{\\beta}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\tilde{y}}= \\boldsymbol{X}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and in order to find the optimal parameters $\\beta_i$ instead of solving the above linear algebra problem, we define a function which gives a measure of the spread between the values $y_i$ (which represent hopefully the exact values) and the parameterized values $\\tilde{y}_i$, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta})=\\frac{1}{n}\\sum_{i=0}^{n-1}\\left(y_i-\\tilde{y}_i\\right)^2=\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)\\right\\},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or using the matrix $\\boldsymbol{X}$ and in a more compact matrix-vector notation as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta})=\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)\\right\\}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This function is one possible way to define the so-called cost function.\n", + "\n", + "\n", + "\n", + "It is also common to define\n", + "the function $C$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta})=\\frac{1}{2n}\\sum_{i=0}^{n-1}\\left(y_i-\\tilde{y}_i\\right)^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "since when taking the first derivative with respect to the unknown parameters $\\beta$, the factor of $2$ cancels out. \n", + "\n", + "The function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{\\beta})=\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)\\right\\},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "can be linked to the variance of the quantity $y_i$ if we interpret the latter as the mean value. \n", + "When linking (see the discussion below) with the maximum likelihood approach below, we will indeed interpret $y_i$ as a mean value" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y_{i}=\\langle y_i \\rangle = \\beta_0x_{i,0}+\\beta_1x_{i,1}+\\beta_2x_{i,2}+\\dots+\\beta_{n-1}x_{i,n-1}+\\epsilon_i,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\langle y_i \\rangle$ is the mean value. Keep in mind also that\n", + "till now we have treated $y_i$ as the exact value. Normally, the\n", + "response (dependent or outcome) variable $y_i$ the outcome of a\n", + "numerical experiment or another type of experiment and is thus only an\n", + "approximation to the true value. It is then always accompanied by an\n", + "error estimate, often limited to a statistical error estimate given by\n", + "the standard deviation discussed earlier. In the discussion here we\n", + "will treat $y_i$ as our exact value for the response variable.\n", + "\n", + "In order to find the parameters $\\beta_i$ we will then minimize the spread of $C(\\boldsymbol{\\beta})$, that is we are going to solve the problem" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in\n", + "{\\mathbb{R}}^{p}}}\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)\\right\\}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In practical terms it means we will require" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial C(\\boldsymbol{\\beta})}{\\partial \\beta_j} = \\frac{\\partial }{\\partial \\beta_j}\\left[ \\frac{1}{n}\\sum_{i=0}^{n-1}\\left(y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}\\right)^2\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which results in" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial C(\\boldsymbol{\\beta})}{\\partial \\beta_j} = -\\frac{2}{n}\\left[ \\sum_{i=0}^{n-1}x_{ij}\\left(y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}\\right)\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or in a matrix-vector form as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial C(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can rewrite" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial C(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{y} = \\boldsymbol{X}^T\\boldsymbol{X}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and if the matrix $\\boldsymbol{X}^T\\boldsymbol{X}$ is invertible we have the solution" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta} =\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We note also that since our design matrix is defined as $\\boldsymbol{X}\\in\n", + "{\\mathbb{R}}^{n\\times p}$, the product $\\boldsymbol{X}^T\\boldsymbol{X} \\in\n", + "{\\mathbb{R}}^{p\\times p}$. In the above case we have that $p \\ll n$,\n", + "in our case $p=5$ meaning that we end up with inverting a small\n", + "$5\\times 5$ matrix. This is a rather common situation, in many cases we end up with low-dimensional\n", + "matrices to invert. The methods discussed here and for many other\n", + "supervised learning algorithms like classification with logistic\n", + "regression or support vector machines, exhibit dimensionalities which\n", + "allow for the usage of direct linear algebra methods such as **LU** decomposition or **Singular Value Decomposition** (SVD) for finding the inverse of the matrix\n", + "$\\boldsymbol{X}^T\\boldsymbol{X}$. \n", + "\n", + "**Small question**: Do you think the example we have at hand here (the nuclear binding energies) can lead to problems in inverting the matrix $\\boldsymbol{X}^T\\boldsymbol{X}$? What kind of problems can we expect? \n", + "\n", + "\n", + "The following matrix and vector relation will be useful here and for the rest of the course. Vectors are always written as boldfaced lower case letters and \n", + "matrices as upper case boldfaced letters." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "4\n", + "8\n", + " \n", + "<\n", + "<\n", + "<\n", + "!\n", + "!\n", + "M\n", + "A\n", + "T\n", + "H\n", + "_\n", + "B\n", + "L\n", + "O\n", + "C\n", + "K" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "4\n", + "9\n", + " \n", + "<\n", + "<\n", + "<\n", + "!\n", + "!\n", + "M\n", + "A\n", + "T\n", + "H\n", + "_\n", + "B\n", + "L\n", + "O\n", + "C\n", + "K" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "5\n", + "0\n", + " \n", + "<\n", + "<\n", + "<\n", + "!\n", + "!\n", + "M\n", + "A\n", + "T\n", + "H\n", + "_\n", + "B\n", + "L\n", + "O\n", + "C\n", + "K" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\log{\\vert\\boldsymbol{A}\\vert}}{\\partial \\boldsymbol{A}} = (\\boldsymbol{A}^{-1})^T.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The residuals $\\boldsymbol{\\epsilon}$ are in turn given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\epsilon} = \\boldsymbol{y}-\\boldsymbol{\\tilde{y}} = \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and with" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)= 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{\\epsilon}=\\boldsymbol{X}^T\\left( \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)= 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "meaning that the solution for $\\boldsymbol{\\beta}$ is the one which minimizes the residuals. Later we will link this with the maximum likelihood approach.\n", + "\n", + "\n", + "Let us now return to our nuclear binding energies and simply code the above equations. \n", + "\n", + "\n", + "It is rather straightforward to implement the matrix inversion and obtain the parameters $\\boldsymbol{\\beta}$. After having defined the matrix $\\boldsymbol{X}$ we simply need to \n", + "write" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# matrix inversion to find beta\n", + "beta = np.linalg.inv(X.T.dot(X)).dot(X.T).dot(Energies)\n", + "# and then make the prediction\n", + "ytilde = X @ beta" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Alternatively, you can use the least squares functionality in **Numpy** as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "fit = np.linalg.lstsq(X, Energies, rcond =None)[0]\n", + "ytildenp = np.dot(fit,X.T)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "And finally we plot our fit with and compare with data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "Masses['Eapprox'] = ytilde\n", + "# Generate a plot comparing the experimental with the fitted values values.\n", + "fig, ax = plt.subplots()\n", + "ax.set_xlabel(r'$A = N + Z$')\n", + "ax.set_ylabel(r'$E_\\mathrm{bind}\\,/\\mathrm{MeV}$')\n", + "ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2,\n", + " label='Ame2016')\n", + "ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m',\n", + " label='Fit')\n", + "ax.legend()\n", + "save_fig(\"Masses2016OLS\")\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can easily test our fit by computing the $R2$ score that we discussed in connection with the functionality of **Scikit-Learn** in the introductory slides.\n", + "Since we are not using **Scikit-Learn** here we can define our own $R2$ function as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "def R2(y_data, y_model):\n", + " return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and we would be using it as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "print(R2(Energies,ytilde))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can easily add our **MSE** score as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "\n", + "print(MSE(Energies,ytilde))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and finally the relative error as" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "def RelativeError(y_data,y_model):\n", + " return abs((y_data-y_model)/y_data)\n", + "print(RelativeError(Energies, ytilde))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### The $\\chi^2$ function\n", + "\n", + "Normally, the response (dependent or outcome) variable $y_i$ is the\n", + "outcome of a numerical experiment or another type of experiment and is\n", + "thus only an approximation to the true value. It is then always\n", + "accompanied by an error estimate, often limited to a statistical error\n", + "estimate given by the standard deviation discussed earlier. In the\n", + "discussion here we will treat $y_i$ as our exact value for the\n", + "response variable.\n", + "\n", + "Introducing the standard deviation $\\sigma_i$ for each measurement\n", + "$y_i$, we define now the $\\chi^2$ function (omitting the $1/n$ term)\n", + "as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\chi^2(\\boldsymbol{\\beta})=\\frac{1}{n}\\sum_{i=0}^{n-1}\\frac{\\left(y_i-\\tilde{y}_i\\right)^2}{\\sigma_i^2}=\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)^T\\frac{1}{\\boldsymbol{\\Sigma^2}}\\left(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}}\\right)\\right\\},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where the matrix $\\boldsymbol{\\Sigma}$ is a diagonal matrix with $\\sigma_i$ as matrix elements. \n", + "\n", + "\n", + "In order to find the parameters $\\beta_i$ we will then minimize the spread of $\\chi^2(\\boldsymbol{\\beta})$ by requiring" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_j} = \\frac{\\partial }{\\partial \\beta_j}\\left[ \\frac{1}{n}\\sum_{i=0}^{n-1}\\left(\\frac{y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}}{\\sigma_i}\\right)^2\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which results in" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_j} = -\\frac{2}{n}\\left[ \\sum_{i=0}^{n-1}\\frac{x_{ij}}{\\sigma_i}\\left(\\frac{y_i-\\beta_0x_{i,0}-\\beta_1x_{i,1}-\\beta_2x_{i,2}-\\dots-\\beta_{n-1}x_{i,n-1}}{\\sigma_i}\\right)\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or in a matrix-vector form as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{A}^T\\left( \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{\\beta}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have defined the matrix $\\boldsymbol{A} =\\boldsymbol{X}/\\boldsymbol{\\Sigma}$ with matrix elements $a_{ij} = x_{ij}/\\sigma_i$ and the vector $\\boldsymbol{b}$ with elements $b_i = y_i/\\sigma_i$. \n", + "\n", + "We can rewrite" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = 0 = \\boldsymbol{A}^T\\left( \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{\\beta}\\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}^T\\boldsymbol{b} = \\boldsymbol{A}^T\\boldsymbol{A}\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and if the matrix $\\boldsymbol{A}^T\\boldsymbol{A}$ is invertible we have the solution" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta} =\\left(\\boldsymbol{A}^T\\boldsymbol{A}\\right)^{-1}\\boldsymbol{A}^T\\boldsymbol{b}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we then introduce the matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{H} = \\left(\\boldsymbol{A}^T\\boldsymbol{A}\\right)^{-1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we have then the following expression for the parameters $\\beta_j$ (the matrix elements of $\\boldsymbol{H}$ are $h_{ij}$)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_j = \\sum_{k=0}^{p-1}h_{jk}\\sum_{i=0}^{n-1}\\frac{y_i}{\\sigma_i}\\frac{x_{ik}}{\\sigma_i} = \\sum_{k=0}^{p-1}h_{jk}\\sum_{i=0}^{n-1}b_ia_{ik}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We state without proof the expression for the uncertainty in the parameters $\\beta_j$ as (we leave this as an exercise)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\sigma^2(\\beta_j) = \\sum_{i=0}^{n-1}\\sigma_i^2\\left( \\frac{\\partial \\beta_j}{\\partial y_i}\\right)^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "resulting in" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\sigma^2(\\beta_j) = \\left(\\sum_{k=0}^{p-1}h_{jk}\\sum_{i=0}^{n-1}a_{ik}\\right)\\left(\\sum_{l=0}^{p-1}h_{jl}\\sum_{m=0}^{n-1}a_{ml}\\right) = h_{jj}!\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The first step here is to approximate the function $y$ with a first-order polynomial, that is we write" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y=y(x) \\rightarrow y(x_i) \\approx \\beta_0+\\beta_1 x_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "By computing the derivatives of $\\chi^2$ with respect to $\\beta_0$ and $\\beta_1$ show that these are given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_0} = -2\\left[ \\frac{1}{n}\\sum_{i=0}^{n-1}\\left(\\frac{y_i-\\beta_0-\\beta_1x_{i}}{\\sigma_i^2}\\right)\\right]=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\chi^2(\\boldsymbol{\\beta})}{\\partial \\beta_1} = -\\frac{2}{n}\\left[ \\sum_{i=0}^{n-1}x_i\\left(\\frac{y_i-\\beta_0-\\beta_1x_{i}}{\\sigma_i^2}\\right)\\right]=0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "For a linear fit (a first-order polynomial) we don't need to invert a matrix!! \n", + "Defining" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma = \\sum_{i=0}^{n-1}\\frac{1}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma_x = \\sum_{i=0}^{n-1}\\frac{x_{i}}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma_y = \\sum_{i=0}^{n-1}\\left(\\frac{y_i}{\\sigma_i^2}\\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma_{xx} = \\sum_{i=0}^{n-1}\\frac{x_ix_{i}}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\gamma_{xy} = \\sum_{i=0}^{n-1}\\frac{y_ix_{i}}{\\sigma_i^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_0 = \\frac{\\gamma_{xx}\\gamma_y-\\gamma_x\\gamma_y}{\\gamma\\gamma_{xx}-\\gamma_x^2},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_1 = \\frac{\\gamma_{xy}\\gamma-\\gamma_x\\gamma_y}{\\gamma\\gamma_{xx}-\\gamma_x^2}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This approach (different linear and non-linear regression) suffers\n", + "often from both being underdetermined and overdetermined in the\n", + "unknown coefficients $\\beta_i$. A better approach is to use the\n", + "Singular Value Decomposition (SVD) method discussed below. Or using\n", + "Lasso and Ridge regression. See below.\n", + "\n", + "\n", + "### Fitting an Equation of State for Dense Nuclear Matter\n", + "\n", + "Before we continue, let us introduce yet another example. We are going to fit the\n", + "nuclear equation of state using results from many-body calculations.\n", + "The equation of state we have made available here, as function of\n", + "density, has been derived using modern nucleon-nucleon potentials with\n", + "[the addition of three-body\n", + "forces](https://www.sciencedirect.com/science/article/pii/S0370157399001106). This\n", + "time the file is presented as a standard **csv** file.\n", + "\n", + "The beginning of the Python code here is similar to what you have seen\n", + "before, with the same initializations and declarations. We use also\n", + "**pandas** again, rather extensively in order to organize our data.\n", + "\n", + "The difference now is that we use **Scikit-Learn's** regression tools\n", + "instead of our own matrix inversion implementation. Furthermore, we\n", + "sneak in **Ridge** regression (to be discussed below) which includes a\n", + "hyperparameter $\\lambda$, also to be explained below." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "import matplotlib.pyplot as plt\n", + "import sklearn.linear_model as skl\n", + "from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"EoS.csv\"),'r')\n", + "\n", + "# Read the EoS data as csv file and organize the data into two arrays with density and energies\n", + "EoS = pd.read_csv(infile, names=('Density', 'Energy'))\n", + "EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')\n", + "EoS = EoS.dropna()\n", + "Energies = EoS['Energy']\n", + "Density = EoS['Density']\n", + "# The design matrix now as function of various polytrops\n", + "X = np.zeros((len(Density),4))\n", + "X[:,3] = Density**(4.0/3.0)\n", + "X[:,2] = Density\n", + "X[:,1] = Density**(2.0/3.0)\n", + "X[:,0] = 1\n", + "\n", + "# We use now Scikit-Learn's linear regressor and ridge regressor\n", + "# OLS part\n", + "clf = skl.LinearRegression().fit(X, Energies)\n", + "ytilde = clf.predict(X)\n", + "EoS['Eols'] = ytilde\n", + "# The mean squared error \n", + "print(\"Mean squared error: %.2f\" % mean_squared_error(Energies, ytilde))\n", + "# Explained variance score: 1 is perfect prediction \n", + "print('Variance score: %.2f' % r2_score(Energies, ytilde))\n", + "# Mean absolute error \n", + "print('Mean absolute error: %.2f' % mean_absolute_error(Energies, ytilde))\n", + "print(clf.coef_, clf.intercept_)\n", + "\n", + "# The Ridge regression with a hyperparameter lambda = 0.1\n", + "_lambda = 0.1\n", + "clf_ridge = skl.Ridge(alpha=_lambda).fit(X, Energies)\n", + "yridge = clf_ridge.predict(X)\n", + "EoS['Eridge'] = yridge\n", + "# The mean squared error \n", + "print(\"Mean squared error: %.2f\" % mean_squared_error(Energies, yridge))\n", + "# Explained variance score: 1 is perfect prediction \n", + "print('Variance score: %.2f' % r2_score(Energies, yridge))\n", + "# Mean absolute error \n", + "print('Mean absolute error: %.2f' % mean_absolute_error(Energies, yridge))\n", + "print(clf_ridge.coef_, clf_ridge.intercept_)\n", + "\n", + "fig, ax = plt.subplots()\n", + "ax.set_xlabel(r'$\\rho[\\mathrm{fm}^{-3}]$')\n", + "ax.set_ylabel(r'Energy per particle')\n", + "ax.plot(EoS['Density'], EoS['Energy'], alpha=0.7, lw=2,\n", + " label='Theoretical data')\n", + "ax.plot(EoS['Density'], EoS['Eols'], alpha=0.7, lw=2, c='m',\n", + " label='OLS')\n", + "ax.plot(EoS['Density'], EoS['Eridge'], alpha=0.7, lw=2, c='g',\n", + " label='Ridge $\\lambda = 0.1$')\n", + "ax.legend()\n", + "save_fig(\"EoSfitting\")\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The above simple polynomial in density $\\rho$ gives an excellent fit\n", + "to the data. \n", + "\n", + "We note also that there is a small deviation between the\n", + "standard OLS and the Ridge regression at higher densities. We discuss this in more detail\n", + "below.\n", + "\n", + "\n", + "## Splitting our Data in Training and Test data\n", + "\n", + "It is normal in essentially all Machine Learning studies to split the\n", + "data in a training set and a test set (sometimes also an additional\n", + "validation set). **Scikit-Learn** has an own function for this. There\n", + "is no explicit recipe for how much data should be included as training\n", + "data and say test data. An accepted rule of thumb is to use\n", + "approximately $2/3$ to $4/5$ of the data as training data. We will\n", + "postpone a discussion of this splitting to the end of these notes and\n", + "our discussion of the so-called **bias-variance** tradeoff. Here we\n", + "limit ourselves to repeat the above equation of state fitting example\n", + "but now splitting the data into a training set and a test set." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import train_test_split\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "def R2(y_data, y_model):\n", + " return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)\n", + "def MSE(y_data,y_model):\n", + " n = np.size(y_model)\n", + " return np.sum((y_data-y_model)**2)/n\n", + "\n", + "infile = open(data_path(\"EoS.csv\"),'r')\n", + "\n", + "# Read the EoS data as csv file and organized into two arrays with density and energies\n", + "EoS = pd.read_csv(infile, names=('Density', 'Energy'))\n", + "EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')\n", + "EoS = EoS.dropna()\n", + "Energies = EoS['Energy']\n", + "Density = EoS['Density']\n", + "# The design matrix now as function of various polytrops\n", + "X = np.zeros((len(Density),5))\n", + "X[:,0] = 1\n", + "X[:,1] = Density**(2.0/3.0)\n", + "X[:,2] = Density\n", + "X[:,3] = Density**(4.0/3.0)\n", + "X[:,4] = Density**(5.0/3.0)\n", + "# We split the data in test and training data\n", + "X_train, X_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2)\n", + "# matrix inversion to find beta\n", + "beta = np.linalg.inv(X_train.T.dot(X_train)).dot(X_train.T).dot(y_train)\n", + "# and then make the prediction\n", + "ytilde = X_train @ beta\n", + "print(\"Training R2\")\n", + "print(R2(y_train,ytilde))\n", + "print(\"Training MSE\")\n", + "print(MSE(y_train,ytilde))\n", + "ypredict = X_test @ beta\n", + "print(\"Test R2\")\n", + "print(R2(y_test,ypredict))\n", + "print(\"Test MSE\")\n", + "print(MSE(y_test,ypredict))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## The Boston housing data example\n", + "\n", + "The Boston housing \n", + "data set was originally a part of UCI Machine Learning Repository\n", + "and has been removed now. The data set is now included in **Scikit-Learn**'s \n", + "library. There are 506 samples and 13 feature (predictor) variables\n", + "in this data set. The objective is to predict the value of prices of\n", + "the house using the features (predictors) listed here.\n", + "\n", + "The features/predictors are\n", + "1. CRIM: Per capita crime rate by town\n", + "\n", + "2. ZN: Proportion of residential land zoned for lots over 25000 square feet\n", + "\n", + "3. INDUS: Proportion of non-retail business acres per town\n", + "\n", + "4. CHAS: Charles River dummy variable (= 1 if tract bounds river; 0 otherwise)\n", + "\n", + "5. NOX: Nitric oxide concentration (parts per 10 million)\n", + "\n", + "6. RM: Average number of rooms per dwelling\n", + "\n", + "7. AGE: Proportion of owner-occupied units built prior to 1940\n", + "\n", + "8. DIS: Weighted distances to five Boston employment centers\n", + "\n", + "9. RAD: Index of accessibility to radial highways\n", + "\n", + "10. TAX: Full-value property tax rate per USD10000\n", + "\n", + "11. B: $1000(Bk - 0.63)^2$, where $Bk$ is the proportion of [people of African American descent] by town\n", + "\n", + "12. LSTAT: Percentage of lower status of the population\n", + "\n", + "13. MEDV: Median value of owner-occupied homes in USD 1000s\n", + "\n", + "## Housing data, the code\n", + "We start by importing the libraries" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt \n", + "\n", + "import pandas as pd \n", + "import seaborn as sns" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and load the Boston Housing DataSet from **Scikit-Learn**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from sklearn.datasets import load_boston\n", + "\n", + "boston_dataset = load_boston()\n", + "\n", + "# boston_dataset is a dictionary\n", + "# let's check what it contains\n", + "boston_dataset.keys()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Then we invoke Pandas" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "boston = pd.DataFrame(boston_dataset.data, columns=boston_dataset.feature_names)\n", + "boston.head()\n", + "boston['MEDV'] = boston_dataset.target" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and preprocess the data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# check for missing values in all the columns\n", + "boston.isnull().sum()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can then visualize the data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# set the size of the figure\n", + "sns.set(rc={'figure.figsize':(11.7,8.27)})\n", + "\n", + "# plot a histogram showing the distribution of the target values\n", + "sns.distplot(boston['MEDV'], bins=30)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "It is now useful to look at the correlation matrix" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# compute the pair wise correlation for all columns \n", + "correlation_matrix = boston.corr().round(2)\n", + "# use the heatmap function from seaborn to plot the correlation matrix\n", + "# annot = True to print the values inside the square\n", + "sns.heatmap(data=correlation_matrix, annot=True)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "From the above coorelation plot we can see that **MEDV** is strongly correlated to **LSTAT** and **RM**. We see also that **RAD** and **TAX** are stronly correlated, but we don't include this in our features together to avoid multi-colinearity" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "plt.figure(figsize=(20, 5))\n", + "\n", + "features = ['LSTAT', 'RM']\n", + "target = boston['MEDV']\n", + "\n", + "for i, col in enumerate(features):\n", + " plt.subplot(1, len(features) , i+1)\n", + " x = boston[col]\n", + " y = target\n", + " plt.scatter(x, y, marker='o')\n", + " plt.title(col)\n", + " plt.xlabel(col)\n", + " plt.ylabel('MEDV')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Now we start training our model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "X = pd.DataFrame(np.c_[boston['LSTAT'], boston['RM']], columns = ['LSTAT','RM'])\n", + "Y = boston['MEDV']" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We split the data into training and test sets" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from sklearn.model_selection import train_test_split\n", + "\n", + "# splits the training and test data set in 80% : 20%\n", + "# assign random_state to any value.This ensures consistency.\n", + "X_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size = 0.2, random_state=5)\n", + "print(X_train.shape)\n", + "print(X_test.shape)\n", + "print(Y_train.shape)\n", + "print(Y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Then we use the linear regression functionality from **Scikit-Learn**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from sklearn.linear_model import LinearRegression\n", + "from sklearn.metrics import mean_squared_error, r2_score\n", + "\n", + "lin_model = LinearRegression()\n", + "lin_model.fit(X_train, Y_train)\n", + "\n", + "# model evaluation for training set\n", + "\n", + "y_train_predict = lin_model.predict(X_train)\n", + "rmse = (np.sqrt(mean_squared_error(Y_train, y_train_predict)))\n", + "r2 = r2_score(Y_train, y_train_predict)\n", + "\n", + "print(\"The model performance for training set\")\n", + "print(\"--------------------------------------\")\n", + "print('RMSE is {}'.format(rmse))\n", + "print('R2 score is {}'.format(r2))\n", + "print(\"\\n\")\n", + "\n", + "# model evaluation for testing set\n", + "\n", + "y_test_predict = lin_model.predict(X_test)\n", + "# root mean square error of the model\n", + "rmse = (np.sqrt(mean_squared_error(Y_test, y_test_predict)))\n", + "\n", + "# r-squared score of the model\n", + "r2 = r2_score(Y_test, y_test_predict)\n", + "\n", + "print(\"The model performance for testing set\")\n", + "print(\"--------------------------------------\")\n", + "print('RMSE is {}'.format(rmse))\n", + "print('R2 score is {}'.format(r2))" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# plotting the y_test vs y_pred\n", + "# ideally should have been a straight line\n", + "plt.scatter(Y_test, y_test_predict)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Reducing the number of degrees of freedom, overarching view\n", + "\n", + "Many Machine Learning problems involve thousands or even millions of\n", + "features for each training instance. Not only does this make training\n", + "extremely slow, it can also make it much harder to find a good\n", + "solution, as we will see. This problem is often referred to as the\n", + "curse of dimensionality. Fortunately, in real-world problems, it is\n", + "often possible to reduce the number of features considerably, turning\n", + "an intractable problem into a tractable one.\n", + "\n", + "Later we will discuss some of the most popular dimensionality reduction\n", + "techniques: the principal component analysis (PCA), Kernel PCA, and\n", + "Locally Linear Embedding (LLE). \n", + "\n", + "\n", + "Principal component analysis and its various variants deal with the\n", + "problem of fitting a low-dimensional [affine\n", + "subspace](https://en.wikipedia.org/wiki/Affine_space) to a set of of\n", + "data points in a high-dimensional space. With its family of methods it\n", + "is one of the most used tools in data modeling, compression and\n", + "visualization.\n", + "\n", + "\n", + "Before we proceed however, we will discuss how to preprocess our\n", + "data. Till now and in connection with our previous examples we have\n", + "not met so many cases where we are too sensitive to the scaling of our\n", + "data. Normally the data may need a rescaling and/or may be sensitive\n", + "to extreme values. Scaling the data renders our inputs much more\n", + "suitable for the algorithms we want to employ.\n", + "\n", + "**Scikit-Learn** has several functions which allow us to rescale the\n", + "data, normally resulting in much better results in terms of various\n", + "accuracy scores. The **StandardScaler** function in **Scikit-Learn**\n", + "ensures that for each feature/predictor we study the mean value is\n", + "zero and the variance is one (every column in the design/feature\n", + "matrix). This scaling has the drawback that it does not ensure that\n", + "we have a particular maximum or minimum in our data set. Another\n", + "function included in **Scikit-Learn** is the **MinMaxScaler** which\n", + "ensures that all features are exactly between $0$ and $1$. The\n", + "\n", + "\n", + "The **Normalizer** scales each data\n", + "point such that the feature vector has a euclidean length of one. In other words, it\n", + "projects a data point on the circle (or sphere in the case of higher dimensions) with a\n", + "radius of 1. This means every data point is scaled by a different number (by the\n", + "inverse of it’s length).\n", + "This normalization is often used when only the direction (or angle) of the data matters,\n", + "not the length of the feature vector.\n", + "\n", + "The **RobustScaler** works similarly to the StandardScaler in that it\n", + "ensures statistical properties for each feature that guarantee that\n", + "they are on the same scale. However, the RobustScaler uses the median\n", + "and quartiles, instead of mean and variance. This makes the\n", + "RobustScaler ignore data points that are very different from the rest\n", + "(like measurement errors). These odd data points are also called\n", + "outliers, and might often lead to trouble for other scaling\n", + "techniques.\n", + "\n", + "\n", + "### Simple preprocessing examples, Franke function and regression" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "import sklearn.linear_model as skl\n", + "from sklearn.metrics import mean_squared_error\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.preprocessing import MinMaxScaler, StandardScaler, Normalizer\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "\n", + "def FrankeFunction(x,y):\n", + "\tterm1 = 0.75*np.exp(-(0.25*(9*x-2)**2) - 0.25*((9*y-2)**2))\n", + "\tterm2 = 0.75*np.exp(-((9*x+1)**2)/49.0 - 0.1*(9*y+1))\n", + "\tterm3 = 0.5*np.exp(-(9*x-7)**2/4.0 - 0.25*((9*y-3)**2))\n", + "\tterm4 = -0.2*np.exp(-(9*x-4)**2 - (9*y-7)**2)\n", + "\treturn term1 + term2 + term3 + term4\n", + "\n", + "\n", + "def create_X(x, y, n ):\n", + "\tif len(x.shape) > 1:\n", + "\t\tx = np.ravel(x)\n", + "\t\ty = np.ravel(y)\n", + "\n", + "\tN = len(x)\n", + "\tl = int((n+1)*(n+2)/2)\t\t# Number of elements in beta\n", + "\tX = np.ones((N,l))\n", + "\n", + "\tfor i in range(1,n+1):\n", + "\t\tq = int((i)*(i+1)/2)\n", + "\t\tfor k in range(i+1):\n", + "\t\t\tX[:,q+k] = (x**(i-k))*(y**k)\n", + "\n", + "\treturn X\n", + "\n", + "\n", + "# Making meshgrid of datapoints and compute Franke's function\n", + "n = 5\n", + "N = 1000\n", + "x = np.sort(np.random.uniform(0, 1, N))\n", + "y = np.sort(np.random.uniform(0, 1, N))\n", + "z = FrankeFunction(x, y)\n", + "X = create_X(x, y, n=n) \n", + "# split in training and test data\n", + "X_train, X_test, y_train, y_test = train_test_split(X,z,test_size=0.2)\n", + "\n", + "\n", + "clf = skl.LinearRegression().fit(X_train, y_train)\n", + "\n", + "# The mean squared error and R2 score\n", + "print(\"MSE before scaling: {:.2f}\".format(mean_squared_error(clf.predict(X_test), y_test)))\n", + "print(\"R2 score before scaling {:.2f}\".format(clf.score(X_test,y_test)))\n", + "\n", + "scaler = StandardScaler()\n", + "scaler.fit(X_train)\n", + "X_train_scaled = scaler.transform(X_train)\n", + "X_test_scaled = scaler.transform(X_test)\n", + "\n", + "print(\"Feature min values before scaling:\\n {}\".format(X_train.min(axis=0)))\n", + "print(\"Feature max values before scaling:\\n {}\".format(X_train.max(axis=0)))\n", + "\n", + "print(\"Feature min values after scaling:\\n {}\".format(X_train_scaled.min(axis=0)))\n", + "print(\"Feature max values after scaling:\\n {}\".format(X_train_scaled.max(axis=0)))\n", + "\n", + "clf = skl.LinearRegression().fit(X_train_scaled, y_train)\n", + "\n", + "\n", + "print(\"MSE after scaling: {:.2f}\".format(mean_squared_error(clf.predict(X_test_scaled), y_test)))\n", + "print(\"R2 score for scaled data: {:.2f}\".format(clf.score(X_test_scaled,y_test)))" + ] + } + ], + "metadata": { + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} \ No newline at end of file diff --git a/doc/LectureNotes/_build/jupyter_execute/chapter1.py b/doc/LectureNotes/_build/jupyter_execute/chapter1.py new file mode 100644 index 000000000..310ea62e5 --- /dev/null +++ b/doc/LectureNotes/_build/jupyter_execute/chapter1.py @@ -0,0 +1,2454 @@ +# Linear Regression, basic Elements + +[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK3155/h20/forelesningsvideoer/LectureAug21.mp4?vrtx=view-as-webpage) + + +## Introduction + + + + + +Our emphasis throughout this series of lectures +is on understanding the mathematical aspects of +different algorithms used in the fields of data analysis and machine learning. + +However, where possible we will emphasize the +importance of using available software. We start thus with a hands-on +and top-down approach to machine learning. The aim is thus to start with +relevant data or data we have produced +and use these to introduce statistical data analysis +concepts and machine learning algorithms before we delve into the +algorithms themselves. The examples we will use in the beginning, start with simple +polynomials with random noise added. We will use the Python +software package [Scikit-Learn](http://scikit-learn.org/stable/) and +introduce various machine learning algorithms to make fits of +the data and predictions. We move thereafter to more interesting +cases such as data from say experiments (below we will look at experimental nuclear binding energies as an example). +These are examples where we can easily set up the data and +then use machine learning algorithms included in for example +**Scikit-Learn**. + +These examples will serve us the purpose of getting +started. Furthermore, they allow us to catch more than two birds with +a stone. They will allow us to bring in some programming specific +topics and tools as well as showing the power of various Python +libraries for machine learning and statistical data analysis. + +Here, we will mainly focus on two +specific Python packages for Machine Learning, Scikit-Learn and +Tensorflow (see below for links etc). Moreover, the examples we +introduce will serve as inputs to many of our discussions later, as +well as allowing you to set up models and produce your own data and +get started with programming. + + + +## What is Machine Learning? + +Statistics, data science and machine learning form important fields of +research in modern science. They describe how to learn and make +predictions from data, as well as allowing us to extract important +correlations about physical process and the underlying laws of motion +in large data sets. The latter, big data sets, appear frequently in +essentially all disciplines, from the traditional Science, Technology, +Mathematics and Engineering fields to Life Science, Law, education +research, the Humanities and the Social Sciences. + +It has become more +and more common to see research projects on big data in for example +the Social Sciences where extracting patterns from complicated survey +data is one of many research directions. Having a solid grasp of data +analysis and machine learning is thus becoming central to scientific +computing in many fields, and competences and skills within the fields +of machine learning and scientific computing are nowadays strongly +requested by many potential employers. The latter cannot be +overstated, familiarity with machine learning has almost become a +prerequisite for many of the most exciting employment opportunities, +whether they are in bioinformatics, life science, physics or finance, +in the private or the public sector. This author has had several +students or met students who have been hired recently based on their +skills and competences in scientific computing and data science, often +with marginal knowledge of machine learning. + +Machine learning is a subfield of computer science, and is closely +related to computational statistics. It evolved from the study of +pattern recognition in artificial intelligence (AI) research, and has +made contributions to AI tasks like computer vision, natural language +processing and speech recognition. Many of the methods we will study are also +strongly rooted in basic mathematics and physics research. + +Ideally, machine learning represents the science of giving computers +the ability to learn without being explicitly programmed. The idea is +that there exist generic algorithms which can be used to find patterns +in a broad class of data sets without having to write code +specifically for each problem. The algorithm will build its own logic +based on the data. You should however always keep in mind that +machines and algorithms are to a large extent developed by humans. The +insights and knowledge we have about a specific system, play a central +role when we develop a specific machine learning algorithm. + +Machine learning is an extremely rich field, in spite of its young +age. The increases we have seen during the last three decades in +computational capabilities have been followed by developments of +methods and techniques for analyzing and handling large date sets, +relying heavily on statistics, computer science and mathematics. The +field is rather new and developing rapidly. Popular software packages +written in Python for machine learning like +[Scikit-learn](http://scikit-learn.org/stable/), +[Tensorflow](https://www.tensorflow.org/), +[PyTorch](http://pytorch.org/) and [Keras](https://keras.io/), all +freely available at their respective GitHub sites, encompass +communities of developers in the thousands or more. And the number of +code developers and contributors keeps increasing. Not all the +algorithms and methods can be given a rigorous mathematical +justification, opening up thereby large rooms for experimenting and +trial and error and thereby exciting new developments. However, a +solid command of linear algebra, multivariate theory, probability +theory, statistical data analysis, understanding errors and Monte +Carlo methods are central elements in a proper understanding of many +of algorithms and methods we will discuss. + + + +The approaches to machine learning are many, but are often split into +two main categories. In *supervised learning* we know the answer to a +problem, and let the computer deduce the logic behind it. On the other +hand, *unsupervised learning* is a method for finding patterns and +relationship in data sets without any prior knowledge of the system. +Some authours also operate with a third category, namely +*reinforcement learning*. This is a paradigm of learning inspired by +behavioral psychology, where learning is achieved by trial-and-error, +solely from rewards and punishment. + +Another way to categorize machine learning tasks is to consider the +desired output of a system. Some of the most common tasks are: + + * Classification: Outputs are divided into two or more classes. The goal is to produce a model that assigns inputs into one of these classes. An example is to identify digits based on pictures of hand-written ones. Classification is typically supervised learning. + + * Regression: Finding a functional relationship between an input data set and a reference data set. The goal is to construct a function that maps input data to continuous output values. + + * Clustering: Data are divided into groups with certain common traits, without knowing the different groups beforehand. It is thus a form of unsupervised learning. + +The methods we cover have three main topics in common, irrespective of +whether we deal with supervised or unsupervised learning. The first +ingredient is normally our data set (which can be subdivided into +training and test data), the second item is a model which is normally a +function of some parameters. The model reflects our knowledge of the system (or lack thereof). As an example, if we know that our data show a behavior similar to what would be predicted by a polynomial, fitting our data to a polynomial of some degree would then determin our model. + +The last ingredient is a so-called **cost** +function which allows us to present an estimate on how good our model +is in reproducing the data it is supposed to train. +At the heart of basically all ML algorithms there are so-called minimization algorithms, often we end up with various variants of **gradient** methods. + + + + + + + +## Software and needed installations + +We will make extensive use of Python as programming language and its +myriad of available libraries. You will find +Jupyter notebooks invaluable in your work. You can run **R** +codes in the Jupyter/IPython notebooks, with the immediate benefit of +visualizing your data. You can also use compiled languages like C++, +Rust, Julia, Fortran etc if you prefer. The focus in these lectures will be +on Python. + + +If you have Python installed (we strongly recommend Python3) and you feel +pretty familiar with installing different packages, we recommend that +you install the following Python packages via **pip** as + +1. pip install numpy scipy matplotlib ipython scikit-learn mglearn sympy pandas pillow + +For Python3, replace **pip** with **pip3**. + +For OSX users we recommend, after having installed Xcode, to +install **brew**. Brew allows for a seamless installation of additional +software via for example + +1. brew install python3 + +For Linux users, with its variety of distributions like for example the widely popular Ubuntu distribution, +you can use **pip** as well and simply install Python as + +1. sudo apt-get install python3 (or python for pyhton2.7) + +etc etc. + + + +## Python installers + +If you don't want to perform these operations separately and venture +into the hassle of exploring how to set up dependencies and paths, we +recommend two widely used distrubutions which set up all relevant +dependencies for Python, namely + +* [Anaconda](https://docs.anaconda.com/), + +which is an open source +distribution of the Python and R programming languages for large-scale +data processing, predictive analytics, and scientific computing, that +aims to simplify package management and deployment. Package versions +are managed by the package management system **conda**. + +* [Enthought canopy](https://www.enthought.com/product/canopy/) + +is a Python +distribution for scientific and analytic computing distribution and +analysis environment, available for free and under a commercial +license. + +Furthermore, [Google's Colab](https://colab.research.google.com/notebooks/welcome.ipynb) is a free Jupyter notebook environment that requires +no setup and runs entirely in the cloud. Try it out! + + +## Useful Python libraries +Here we list several useful Python libraries we strongly recommend (if you use anaconda many of these are already there) + +* [NumPy](https://www.numpy.org/) is a highly popular library for large, multi-dimensional arrays and matrices, along with a large collection of high-level mathematical functions to operate on these arrays + +* [The pandas](https://pandas.pydata.org/) library provides high-performance, easy-to-use data structures and data analysis tools + +* [Xarray](http://xarray.pydata.org/en/stable/) is a Python package that makes working with labelled multi-dimensional arrays simple, efficient, and fun! + +* [Scipy](https://www.scipy.org/) (pronounced “Sigh Pie”) is a Python-based ecosystem of open-source software for mathematics, science, and engineering. + +* [Matplotlib](https://matplotlib.org/) is a Python 2D plotting library which produces publication quality figures in a variety of hardcopy formats and interactive environments across platforms. + +* [Autograd](https://github.com/HIPS/autograd) can automatically differentiate native Python and Numpy code. It can handle a large subset of Python's features, including loops, ifs, recursion and closures, and it can even take derivatives of derivatives of derivatives + +* [SymPy](https://www.sympy.org/en/index.html) is a Python library for symbolic mathematics. + +* [scikit-learn](https://scikit-learn.org/stable/) has simple and efficient tools for machine learning, data mining and data analysis + +* [TensorFlow](https://www.tensorflow.org/) is a Python library for fast numerical computing created and released by Google + +* [Keras](https://keras.io/) is a high-level neural networks API, written in Python and capable of running on top of TensorFlow, CNTK, or Theano + +* And many more such as [pytorch](https://pytorch.org/), [Theano](https://pypi.org/project/Theano/) etc + +## Installing R, C++, cython or Julia + +You will also find it convenient to utilize **R**. We will mainly +use Python during our lectures and in various projects and exercises. +Those of you +already familiar with **R** should feel free to continue using **R**, keeping +however an eye on the parallel Python set ups. Similarly, if you are a +Python afecionado, feel free to explore **R** as well. Jupyter/Ipython +notebook allows you to run **R** codes interactively in your +browser. The software library **R** is really tailored for statistical data analysis +and allows for an easy usage of the tools and algorithms we will discuss in these +lectures. + +To install **R** with Jupyter notebook +[follow the link here](https://mpacer.org/maths/r-kernel-for-ipython-notebook) + + + + +## Installing R, C++, cython, Numba etc + + +For the C++ aficionados, Jupyter/IPython notebook allows you also to +install C++ and run codes written in this language interactively in +the browser. Since we will emphasize writing many of the algorithms +yourself, you can thus opt for either Python or C++ (or Fortran or other compiled languages) as programming +languages. + +To add more entropy, **cython** can also be used when running your +notebooks. It means that Python with the jupyter notebook +setup allows you to integrate widely popular softwares and tools for +scientific computing. Similarly, the +[Numba Python package](https://numba.pydata.org/) delivers increased performance +capabilities with minimal rewrites of your codes. With its +versatility, including symbolic operations, Python offers a unique +computational environment. Your jupyter notebook can easily be +converted into a nicely rendered **PDF** file or a Latex file for +further processing. For example, convert to latex as + + pycod jupyter nbconvert filename.ipynb --to latex + + +And to add more versatility, the Python package [SymPy](http://www.sympy.org/en/index.html) is a Python library for symbolic mathematics. It aims to become a full-featured computer algebra system (CAS) and is entirely written in Python. + +Finally, if you wish to use the light mark-up language +[doconce](https://github.com/hplgit/doconce) you can convert a standard ascii text file into various HTML +formats, ipython notebooks, latex files, pdf files etc with minimal edits. These lectures were generated using **doconce**. + + + +## Numpy examples and Important Matrix and vector handling packages + +There are several central software libraries for linear algebra and eigenvalue problems. Several of the more +popular ones have been wrapped into ofter software packages like those from the widely used text **Numerical Recipes**. The original source codes in many of the available packages are often taken from the widely used +software package LAPACK, which follows two other popular packages +developed in the 1970s, namely EISPACK and LINPACK. We describe them shortly here. + + * LINPACK: package for linear equations and least square problems. + + * LAPACK:package for solving symmetric, unsymmetric and generalized eigenvalue problems. From LAPACK's website it is possible to download for free all source codes from this library. Both C/C++ and Fortran versions are available. + + * BLAS (I, II and III): (Basic Linear Algebra Subprograms) are routines that provide standard building blocks for performing basic vector and matrix operations. Blas I is vector operations, II vector-matrix operations and III matrix-matrix operations. Highly parallelized and efficient codes, all available for download from . + +## Basic Matrix Features + +Matrix properties reminder + +$$ +\mathbf{A} = + \begin{bmatrix} a_{11} & a_{12} & a_{13} & a_{14} \\ + a_{21} & a_{22} & a_{23} & a_{24} \\ + a_{31} & a_{32} & a_{33} & a_{34} \\ + a_{41} & a_{42} & a_{43} & a_{44} + \end{bmatrix}\qquad +\mathbf{I} = + \begin{bmatrix} 1 & 0 & 0 & 0 \\ + 0 & 1 & 0 & 0 \\ + 0 & 0 & 1 & 0 \\ + 0 & 0 & 0 & 1 + \end{bmatrix} +$$ + +The inverse of a matrix is defined by + +$$ +\mathbf{A}^{-1} \cdot \mathbf{A} = I +$$ + + + + + + + + + + + + +
Relations Name matrix elements
$A = A^{T}$ symmetric $a_{ij} = a_{ji}$
$A = \left (A^{T} \right )^{-1}$ real orthogonal $\sum_k a_{ik} a_{jk} = \sum_k a_{ki} a_{kj} = \delta_{ij}$
$A = A^{ * }$ real matrix $a_{ij} = a_{ij}^{ * }$
$A = A^{\dagger}$ hermitian $a_{ij} = a_{ji}^{ * }$
$A = \left (A^{\dagger} \right )^{-1}$ unitary $\sum_k a_{ik} a_{jk}^{ * } = \sum_k a_{ki}^{ * } a_{kj} = \delta_{ij}$
+ + +### Some famous Matrices + + * Diagonal if $a_{ij}=0$ for $i\ne j$ + + * Upper triangular if $a_{ij}=0$ for $i > j$ + + * Lower triangular if $a_{ij}=0$ for $i < j$ + + * Upper Hessenberg if $a_{ij}=0$ for $i > j+1$ + + * Lower Hessenberg if $a_{ij}=0$ for $i < j+1$ + + * Tridiagonal if $a_{ij}=0$ for $|i -j| > 1$ + + * Lower banded with bandwidth $p$: $a_{ij}=0$ for $i > j+p$ + + * Upper banded with bandwidth $p$: $a_{ij}=0$ for $i < j+p$ + + * Banded, block upper triangular, block lower triangular.... + +### More Basic Matrix Features + +Some Equivalent Statements +For an $N\times N$ matrix $\mathbf{A}$ the following properties are all equivalent + + * If the inverse of $\mathbf{A}$ exists, $\mathbf{A}$ is nonsingular. + + * The equation $\mathbf{Ax}=0$ implies $\mathbf{x}=0$. + + * The rows of $\mathbf{A}$ form a basis of $R^N$. + + * The columns of $\mathbf{A}$ form a basis of $R^N$. + + * $\mathbf{A}$ is a product of elementary matrices. + + * $0$ is not eigenvalue of $\mathbf{A}$. + +## Numpy and arrays +[Numpy](http://www.numpy.org/) provides an easy way to handle arrays in Python. The standard way to import this library is as + +import numpy as np + +Here follows a simple example where we set up an array of ten elements, all determined by random numbers drawn according to the normal distribution, + +n = 10 +x = np.random.normal(size=n) +print(x) + +We defined a vector $x$ with $n=10$ elements with its values given by the Normal distribution $N(0,1)$. +Another alternative is to declare a vector as follows + +import numpy as np +x = np.array([1, 2, 3]) +print(x) + +Here we have defined a vector with three elements, with $x_0=1$, $x_1=2$ and $x_2=3$. Note that both Python and C++ +start numbering array elements from $0$ and on. This means that a vector with $n$ elements has a sequence of entities $x_0, x_1, x_2, \dots, x_{n-1}$. We could also let (recommended) Numpy to compute the logarithms of a specific array as + +import numpy as np +x = np.log(np.array([4, 7, 8])) +print(x) + +In the last example we used Numpy's unary function $np.log$. This function is +highly tuned to compute array elements since the code is vectorized +and does not require looping. We normaly recommend that you use the +Numpy intrinsic functions instead of the corresponding **log** function +from Python's **math** module. The looping is done explicitely by the +**np.log** function. The alternative, and slower way to compute the +logarithms of a vector would be to write + +import numpy as np +from math import log +x = np.array([4, 7, 8]) +for i in range(0, len(x)): + x[i] = log(x[i]) +print(x) + +We note that our code is much longer already and we need to import the **log** function from the **math** module. +The attentive reader will also notice that the output is $[1, 1, 2]$. Python interprets automagically our numbers as integers (like the **automatic** keyword in C++). To change this we could define our array elements to be double precision numbers as + +import numpy as np +x = np.log(np.array([4, 7, 8], dtype = np.float64)) +print(x) + +or simply write them as double precision numbers (Python uses 64 bits as default for floating point type variables), that is + +import numpy as np +x = np.log(np.array([4.0, 7.0, 8.0]) +print(x) + +To check the number of bytes (remember that one byte contains eight bits for double precision variables), you can use simple use the **itemsize** functionality (the array $x$ is actually an object which inherits the functionalities defined in Numpy) as + +import numpy as np +x = np.log(np.array([4.0, 7.0, 8.0]) +print(x.itemsize) + +## Matrices in Python + +Having defined vectors, we are now ready to try out matrices. We can +define a $3 \times 3 $ real matrix $\hat{A}$ as (recall that we user +lowercase letters for vectors and uppercase letters for matrices) + +import numpy as np +A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ])) +print(A) + +If we use the **shape** function we would get $(3, 3)$ as output, that is verifying that our matrix is a $3\times 3$ matrix. We can slice the matrix and print for example the first column (Python organized matrix elements in a row-major order, see below) as + +import numpy as np +A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ])) +# print the first column, row-major order and elements start with 0 +print(A[:,0]) + +We can continue this was by printing out other columns or rows. The example here prints out the second column + +import numpy as np +A = np.log(np.array([ [4.0, 7.0, 8.0], [3.0, 10.0, 11.0], [4.0, 5.0, 7.0] ])) +# print the first column, row-major order and elements start with 0 +print(A[1,:]) + +Numpy contains many other functionalities that allow us to slice, subdivide etc etc arrays. We strongly recommend that you look up the [Numpy website for more details](http://www.numpy.org/). Useful functions when defining a matrix are the **np.zeros** function which declares a matrix of a given dimension and sets all elements to zero + +import numpy as np +n = 10 +# define a matrix of dimension 10 x 10 and set all elements to zero +A = np.zeros( (n, n) ) +print(A) + +or initializing all elements to + +import numpy as np +n = 10 +# define a matrix of dimension 10 x 10 and set all elements to one +A = np.ones( (n, n) ) +print(A) + +or as unitarily distributed random numbers (see the material on random number generators in the statistics part) + +import numpy as np +n = 10 +# define a matrix of dimension 10 x 10 and set all elements to random numbers with x \in [0, 1] +A = np.random.rand(n, n) +print(A) + +As we will see throughout these lectures, there are several extremely useful functionalities in Numpy. +As an example, consider the discussion of the covariance matrix. Suppose we have defined three vectors +$\hat{x}, \hat{y}, \hat{z}$ with $n$ elements each. The covariance matrix is defined as + +$$ +\hat{\Sigma} = \begin{bmatrix} \sigma_{xx} & \sigma_{xy} & \sigma_{xz} \\ + \sigma_{yx} & \sigma_{yy} & \sigma_{yz} \\ + \sigma_{zx} & \sigma_{zy} & \sigma_{zz} + \end{bmatrix}, +$$ + +where for example + +$$ +\sigma_{xy} =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +$$ + +The Numpy function **np.cov** calculates the covariance elements using the factor $1/(n-1)$ instead of $1/n$ since it assumes we do not have the exact mean values. +The following simple function uses the **np.vstack** function which takes each vector of dimension $1\times n$ and produces a $3\times n$ matrix $\hat{W}$ + +$$ +\hat{W} = \begin{bmatrix} x_0 & y_0 & z_0 \\ + x_1 & y_1 & z_1 \\ + x_2 & y_2 & z_2 \\ + \dots & \dots & \dots \\ + x_{n-2} & y_{n-2} & z_{n-2} \\ + x_{n-1} & y_{n-1} & z_{n-1} + \end{bmatrix}, +$$ + +which in turn is converted into into the $3\times 3$ covariance matrix +$\hat{\Sigma}$ via the Numpy function **np.cov()**. We note that we can also calculate +the mean value of each set of samples $\hat{x}$ etc using the Numpy +function **np.mean(x)**. We can also extract the eigenvalues of the +covariance matrix through the **np.linalg.eig()** function. + +# Importing various packages +import numpy as np + +n = 100 +x = np.random.normal(size=n) +print(np.mean(x)) +y = 4+3*x+np.random.normal(size=n) +print(np.mean(y)) +z = x**3+np.random.normal(size=n) +print(np.mean(z)) +W = np.vstack((x, y, z)) +Sigma = np.cov(W) +print(Sigma) +Eigvals, Eigvecs = np.linalg.eig(Sigma) +print(Eigvals) + +%matplotlib inline + +import numpy as np +import matplotlib.pyplot as plt +from scipy import sparse +eye = np.eye(4) +print(eye) +sparse_mtx = sparse.csr_matrix(eye) +print(sparse_mtx) +x = np.linspace(-10,10,100) +y = np.sin(x) +plt.plot(x,y,marker='x') +plt.show() + +## Meet the Pandas + + + + +Another useful Python package is +[pandas](https://pandas.pydata.org/), which is an open source library +providing high-performance, easy-to-use data structures and data +analysis tools for Python. **pandas** stands for panel data, a term borrowed from econometrics and is an efficient library for data analysis with an emphasis on tabular data. +**pandas** has two major classes, the **DataFrame** class with two-dimensional data objects and tabular data organized in columns and the class **Series** with a focus on one-dimensional data objects. Both classes allow you to index data easily as we will see in the examples below. +**pandas** allows you also to perform mathematical operations on the data, spanning from simple reshapings of vectors and matrices to statistical operations. + +The following simple example shows how we can, in an easy way make tables of our data. Here we define a data set which includes names, place of birth and date of birth, and displays the data in an easy to read way. We will see repeated use of **pandas**, in particular in connection with classification of data. + +import pandas as pd +from IPython.display import display +data = {'First Name': ["Frodo", "Bilbo", "Aragorn II", "Samwise"], + 'Last Name': ["Baggins", "Baggins","Elessar","Gamgee"], + 'Place of birth': ["Shire", "Shire", "Eriador", "Shire"], + 'Date of Birth T.A.': [2968, 2890, 2931, 2980] + } +data_pandas = pd.DataFrame(data) +display(data_pandas) + +In the above we have imported **pandas** with the shorthand **pd**, the latter has become the standard way we import **pandas**. We make then a list of various variables +and reorganize the aboves lists into a **DataFrame** and then print out a neat table with specific column labels as *Name*, *place of birth* and *date of birth*. +Displaying these results, we see that the indices are given by the default numbers from zero to three. +**pandas** is extremely flexible and we can easily change the above indices by defining a new type of indexing as + +data_pandas = pd.DataFrame(data,index=['Frodo','Bilbo','Aragorn','Sam']) +display(data_pandas) + +Thereafter we display the content of the row which begins with the index **Aragorn** + +display(data_pandas.loc['Aragorn']) + +We can easily append data to this, for example + +new_hobbit = {'First Name': ["Peregrin"], + 'Last Name': ["Took"], + 'Place of birth': ["Shire"], + 'Date of Birth T.A.': [2990] + } +data_pandas=data_pandas.append(pd.DataFrame(new_hobbit, index=['Pippin'])) +display(data_pandas) + +Here are other examples where we use the **DataFrame** functionality to handle arrays, now with more interesting features for us, namely numbers. We set up a matrix +of dimensionality $10\times 5$ and compute the mean value and standard deviation of each column. Similarly, we can perform mathematial operations like squaring the matrix elements and many other operations. + +import numpy as np +import pandas as pd +from IPython.display import display +np.random.seed(100) +# setting up a 10 x 5 matrix +rows = 10 +cols = 5 +a = np.random.randn(rows,cols) +df = pd.DataFrame(a) +display(df) +print(df.mean()) +print(df.std()) +display(df**2) + +Thereafter we can select specific columns only and plot final results + +df.columns = ['First', 'Second', 'Third', 'Fourth', 'Fifth'] +df.index = np.arange(10) + +display(df) +print(df['Second'].mean() ) +print(df.info()) +print(df.describe()) + +from pylab import plt, mpl +plt.style.use('seaborn') +mpl.rcParams['font.family'] = 'serif' + +df.cumsum().plot(lw=2.0, figsize=(10,6)) +plt.show() + + +df.plot.bar(figsize=(10,6), rot=15) +plt.show() + +We can produce a $4\times 4$ matrix + +b = np.arange(16).reshape((4,4)) +print(b) +df1 = pd.DataFrame(b) +print(df1) + +and many other operations. + +The **Series** class is another important class included in +**pandas**. You can view it as a specialization of **DataFrame** but where +we have just a single column of data. It shares many of the same features as _DataFrame. As with **DataFrame**, +most operations are vectorized, achieving thereby a high performance when dealing with computations of arrays, in particular labeled arrays. +As we will see below it leads also to a very concice code close to the mathematical operations we may be interested in. +For multidimensional arrays, we recommend strongly [xarray](http://xarray.pydata.org/en/stable/). **xarray** has much of the same flexibility as **pandas**, but allows for the extension to higher dimensions than two. We will see examples later of the usage of both **pandas** and **xarray**. + + + + + + +In order to study various Machine Learning algorithms, we need to +access data. Acccessing data is an essential step in all machine +learning algorithms. In particular, setting up the so-called **design +matrix** (to be defined below) is often the first element we need in +order to perform our calculations. To set up the design matrix means +reading (and later, when the calculations are done, writing) data +in various formats, The formats span from reading files from disk, +loading data from databases and interacting with online sources +like web application programming interfaces (APIs). + +In handling various input formats, as discussed above, we will mainly stay with **pandas**, +a Python package which allows us, in a seamless and painless way, to +deal with a multitude of formats, from standard **csv** (comma separated +values) files, via **excel**, **html** to **hdf5** formats. With **pandas** +and the **DataFrame** and **Series** functionalities we are able to convert text data +into the calculational formats we need for a specific algorithm. And our code is going to be +pretty close the basic mathematical expressions. + +Our first data set is going to be a classic from nuclear physics, namely all +available data on binding energies. Don't be intimidated if you are not familiar with nuclear physics. It serves simply as an example here of a data set. + +We will show some of the +strengths of packages like **Scikit-Learn** in fitting nuclear binding energies to +specific functions using linear regression first. Then, as a teaser, we will show you how +you can easily implement other algorithms like decision trees and random forests and neural networks. + +But before we really start with nuclear physics data, let's just look at some simpler polynomial fitting cases, such as, +(don't be offended) fitting straight lines! + + + + +## Simple linear regression model using **scikit-learn** + +We start with perhaps our simplest possible example, using **Scikit-Learn** to perform linear regression analysis on a data set produced by us. + +What follows is a simple Python code where we have defined a function +$y$ in terms of the variable $x$. Both are defined as vectors with $100$ entries. +The numbers in the vector $\hat{x}$ are given +by random numbers generated with a uniform distribution with entries +$x_i \in [0,1]$ (more about probability distribution functions +later). These values are then used to define a function $y(x)$ +(tabulated again as a vector) with a linear dependence on $x$ plus a +random noise added via the normal distribution. + + +The Numpy functions are imported used the **import numpy as np** +statement and the random number generator for the uniform distribution +is called using the function **np.random.rand()**, where we specificy +that we want $100$ random variables. Using Numpy we define +automatically an array with the specified number of elements, $100$ in +our case. With the Numpy function **randn()** we can compute random +numbers with the normal distribution (mean value $\mu$ equal to zero and +variance $\sigma^2$ set to one) and produce the values of $y$ assuming a linear +dependence as function of $x$ + +$$ +y = 2x+N(0,1), +$$ + +where $N(0,1)$ represents random numbers generated by the normal +distribution. From **Scikit-Learn** we import then the +**LinearRegression** functionality and make a prediction $\tilde{y} = +\alpha + \beta x$ using the function **fit(x,y)**. We call the set of +data $(\hat{x},\hat{y})$ for our training data. The Python package +**scikit-learn** has also a functionality which extracts the above +fitting parameters $\alpha$ and $\beta$ (see below). Later we will +distinguish between training data and test data. + +For plotting we use the Python package +[matplotlib](https://matplotlib.org/) which produces publication +quality figures. Feel free to explore the extensive +[gallery](https://matplotlib.org/gallery/index.html) of examples. In +this example we plot our original values of $x$ and $y$ as well as the +prediction **ypredict** ($\tilde{y}$), which attempts at fitting our +data with a straight line. + +The Python code follows here. + +# Importing various packages +import numpy as np +import matplotlib.pyplot as plt +from sklearn.linear_model import LinearRegression + +x = np.random.rand(100,1) +y = 2*x+np.random.randn(100,1) +linreg = LinearRegression() +linreg.fit(x,y) +xnew = np.array([[0],[1]]) +ypredict = linreg.predict(xnew) + +plt.plot(xnew, ypredict, "r-") +plt.plot(x, y ,'ro') +plt.axis([0,1.0,0, 5.0]) +plt.xlabel(r'$x$') +plt.ylabel(r'$y$') +plt.title(r'Simple Linear Regression') +plt.show() + +This example serves several aims. It allows us to demonstrate several +aspects of data analysis and later machine learning algorithms. The +immediate visualization shows that our linear fit is not +impressive. It goes through the data points, but there are many +outliers which are not reproduced by our linear regression. We could +now play around with this small program and change for example the +factor in front of $x$ and the normal distribution. Try to change the +function $y$ to + +$$ +y = 10x+0.01 \times N(0,1), +$$ + +where $x$ is defined as before. Does the fit look better? Indeed, by +reducing the role of the noise given by the normal distribution we see immediately that +our linear prediction seemingly reproduces better the training +set. However, this testing 'by the eye' is obviouly not satisfactory in the +long run. Here we have only defined the training data and our model, and +have not discussed a more rigorous approach to the **cost** function. + +We need more rigorous criteria in defining whether we have succeeded or +not in modeling our training data. You will be surprised to see that +many scientists seldomly venture beyond this 'by the eye' approach. A +standard approach for the *cost* function is the so-called $\chi^2$ +function (a variant of the mean-squared error (MSE)) + +$$ +\chi^2 = \frac{1}{n} +\sum_{i=0}^{n-1}\frac{(y_i-\tilde{y}_i)^2}{\sigma_i^2}, +$$ + +where $\sigma_i^2$ is the variance (to be defined later) of the entry +$y_i$. We may not know the explicit value of $\sigma_i^2$, it serves +however the aim of scaling the equations and make the cost function +dimensionless. + +Minimizing the cost function is a central aspect of +our discussions to come. Finding its minima as function of the model +parameters ($\alpha$ and $\beta$ in our case) will be a recurring +theme in these series of lectures. Essentially all machine learning +algorithms we will discuss center around the minimization of the +chosen cost function. This depends in turn on our specific +model for describing the data, a typical situation in supervised +learning. Automatizing the search for the minima of the cost function is a +central ingredient in all algorithms. Typical methods which are +employed are various variants of **gradient** methods. These will be +discussed in more detail later. Again, you'll be surprised to hear that +many practitioners minimize the above function ''by the eye', popularly dubbed as +'chi by the eye'. That is, change a parameter and see (visually and numerically) that +the $\chi^2$ function becomes smaller. + +There are many ways to define the cost function. A simpler approach is to look at the relative difference between the training data and the predicted data, that is we define +the relative error (why would we prefer the MSE instead of the relative error?) as + +$$ +\epsilon_{\mathrm{relative}}= \frac{\vert \hat{y} -\hat{\tilde{y}}\vert}{\vert \hat{y}\vert}. +$$ + +The squared cost function results in an arithmetic mean-unbiased +estimator, and the absolute-value cost function results in a +median-unbiased estimator (in the one-dimensional case, and a +geometric median-unbiased estimator for the multi-dimensional +case). The squared cost function has the disadvantage that it has the tendency +to be dominated by outliers. + +We can modify easily the above Python code and plot the relative error instead + +import numpy as np +import matplotlib.pyplot as plt +from sklearn.linear_model import LinearRegression + +x = np.random.rand(100,1) +y = 5*x+0.01*np.random.randn(100,1) +linreg = LinearRegression() +linreg.fit(x,y) +ypredict = linreg.predict(x) + +plt.plot(x, np.abs(ypredict-y)/abs(y), "ro") +plt.axis([0,1.0,0.0, 0.5]) +plt.xlabel(r'$x$') +plt.ylabel(r'$\epsilon_{\mathrm{relative}}$') +plt.title(r'Relative error') +plt.show() + +Depending on the parameter in front of the normal distribution, we may +have a small or larger relative error. Try to play around with +different training data sets and study (graphically) the value of the +relative error. + +As mentioned above, **Scikit-Learn** has an impressive functionality. +We can for example extract the values of $\alpha$ and $\beta$ and +their error estimates, or the variance and standard deviation and many +other properties from the statistical data analysis. + +Here we show an +example of the functionality of **Scikit-Learn**. + +import numpy as np +import matplotlib.pyplot as plt +from sklearn.linear_model import LinearRegression +from sklearn.metrics import mean_squared_error, r2_score, mean_squared_log_error, mean_absolute_error + +x = np.random.rand(100,1) +y = 2.0+ 5*x+0.5*np.random.randn(100,1) +linreg = LinearRegression() +linreg.fit(x,y) +ypredict = linreg.predict(x) +print('The intercept alpha: \n', linreg.intercept_) +print('Coefficient beta : \n', linreg.coef_) +# The mean squared error +print("Mean squared error: %.2f" % mean_squared_error(y, ypredict)) +# Explained variance score: 1 is perfect prediction +print('Variance score: %.2f' % r2_score(y, ypredict)) +# Mean squared log error +print('Mean squared log error: %.2f' % mean_squared_log_error(y, ypredict) ) +# Mean absolute error +print('Mean absolute error: %.2f' % mean_absolute_error(y, ypredict)) +plt.plot(x, ypredict, "r-") +plt.plot(x, y ,'ro') +plt.axis([0.0,1.0,1.5, 7.0]) +plt.xlabel(r'$x$') +plt.ylabel(r'$y$') +plt.title(r'Linear Regression fit ') +plt.show() + +The function **coef** gives us the parameter $\beta$ of our fit while **intercept** yields +$\alpha$. Depending on the constant in front of the normal distribution, we get values near or far from $alpha =2$ and $\beta =5$. Try to play around with different parameters in front of the normal distribution. The function **meansquarederror** gives us the mean square error, a risk metric corresponding to the expected value of the squared (quadratic) error or loss defined as + +$$ +MSE(\hat{y},\hat{\tilde{y}}) = \frac{1}{n} +\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2, +$$ + +The smaller the value, the better the fit. Ideally we would like to +have an MSE equal zero. The attentive reader has probably recognized +this function as being similar to the $\chi^2$ function defined above. + +The **r2score** function computes $R^2$, the coefficient of +determination. It provides a measure of how well future samples are +likely to be predicted by the model. Best possible score is 1.0 and it +can be negative (because the model can be arbitrarily worse). A +constant model that always predicts the expected value of $\hat{y}$, +disregarding the input features, would get a $R^2$ score of $0.0$. + +If $\tilde{\hat{y}}_i$ is the predicted value of the $i-th$ sample and $y_i$ is the corresponding true value, then the score $R^2$ is defined as + +$$ +R^2(\hat{y}, \tilde{\hat{y}}) = 1 - \frac{\sum_{i=0}^{n - 1} (y_i - \tilde{y}_i)^2}{\sum_{i=0}^{n - 1} (y_i - \bar{y})^2}, +$$ + +where we have defined the mean value of $\hat{y}$ as + +$$ +\bar{y} = \frac{1}{n} \sum_{i=0}^{n - 1} y_i. +$$ + +Another quantity taht we will meet again in our discussions of regression analysis is + the mean absolute error (MAE), a risk metric corresponding to the expected value of the absolute error loss or what we call the $l1$-norm loss. In our discussion above we presented the relative error. +The MAE is defined as follows + +$$ +\text{MAE}(\hat{y}, \hat{\tilde{y}}) = \frac{1}{n} \sum_{i=0}^{n-1} \left| y_i - \tilde{y}_i \right|. +$$ + +We present the +squared logarithmic (quadratic) error + +$$ +\text{MSLE}(\hat{y}, \hat{\tilde{y}}) = \frac{1}{n} \sum_{i=0}^{n - 1} (\log_e (1 + y_i) - \log_e (1 + \tilde{y}_i) )^2, +$$ + +where $\log_e (x)$ stands for the natural logarithm of $x$. This error +estimate is best to use when targets having exponential growth, such +as population counts, average sales of a commodity over a span of +years etc. + + +Finally, another cost function is the Huber cost function used in robust regression. + +The rationale behind this possible cost function is its reduced +sensitivity to outliers in the data set. In our discussions on +dimensionality reduction and normalization of data we will meet other +ways of dealing with outliers. + +The Huber cost function is defined as + +$$ +H_{\delta}(a)={\begin{cases}{\frac {1}{2}}{a^{2}}&{\text{for }}|a|\leq \delta ,\\\delta (|a|-{\frac {1}{2}}\delta ),&{\text{otherwise.}}\end{cases}}}. +$$ + +Here $a=\boldsymbol{y} - \boldsymbol{\tilde{y}}$. +We will discuss in more +detail these and other functions in the various lectures. We conclude this part with another example. Instead of +a linear $x$-dependence we study now a cubic polynomial and use the polynomial regression analysis tools of scikit-learn. + +import matplotlib.pyplot as plt +import numpy as np +import random +from sklearn.linear_model import Ridge +from sklearn.preprocessing import PolynomialFeatures +from sklearn.pipeline import make_pipeline +from sklearn.linear_model import LinearRegression + +x=np.linspace(0.02,0.98,200) +noise = np.asarray(random.sample((range(200)),200)) +y=x**3*noise +yn=x**3*100 +poly3 = PolynomialFeatures(degree=3) +X = poly3.fit_transform(x[:,np.newaxis]) +clf3 = LinearRegression() +clf3.fit(X,y) + +Xplot=poly3.fit_transform(x[:,np.newaxis]) +poly3_plot=plt.plot(x, clf3.predict(Xplot), label='Cubic Fit') +plt.plot(x,yn, color='red', label="True Cubic") +plt.scatter(x, y, label='Data', color='orange', s=15) +plt.legend() +plt.show() + +def error(a): + for i in y: + err=(y-yn)/yn + return abs(np.sum(err))/len(err) + +print (error(y)) + +Let us now dive into nuclear physics and remind ourselves briefly about some basic features about binding +energies. A basic quantity which can be measured for the ground +states of nuclei is the atomic mass $M(N, Z)$ of the neutral atom with +atomic mass number $A$ and charge $Z$. The number of neutrons is $N$. There are indeed several sophisticated experiments worldwide which allow us to measure this quantity to high precision (parts per million even). + +Atomic masses are usually tabulated in terms of the mass excess defined by + +$$ +\Delta M(N, Z) = M(N, Z) - uA, +$$ + +where $u$ is the Atomic Mass Unit + +$$ +u = M(^{12}\mathrm{C})/12 = 931.4940954(57) \hspace{0.1cm} \mathrm{MeV}/c^2. +$$ + +The nucleon masses are + +$$ +m_p = 1.00727646693(9)u, +$$ + +and + +$$ +m_n = 939.56536(8)\hspace{0.1cm} \mathrm{MeV}/c^2 = 1.0086649156(6)u. +$$ + +In the [2016 mass evaluation of by W.J.Huang, G.Audi, M.Wang, F.G.Kondev, S.Naimi and X.Xu](http://nuclearmasses.org/resources_folder/Wang_2017_Chinese_Phys_C_41_030003.pdf) +there are data on masses and decays of 3437 nuclei. + +The nuclear binding energy is defined as the energy required to break +up a given nucleus into its constituent parts of $N$ neutrons and $Z$ +protons. In terms of the atomic masses $M(N, Z)$ the binding energy is +defined by + +$$ +BE(N, Z) = ZM_H c^2 + Nm_n c^2 - M(N, Z)c^2 , +$$ + +where $M_H$ is the mass of the hydrogen atom and $m_n$ is the mass of the neutron. +In terms of the mass excess the binding energy is given by + +$$ +BE(N, Z) = Z\Delta_H c^2 + N\Delta_n c^2 -\Delta(N, Z)c^2 , +$$ + +where $\Delta_H c^2 = 7.2890$ MeV and $\Delta_n c^2 = 8.0713$ MeV. + + +A popular and physically intuitive model which can be used to parametrize +the experimental binding energies as function of $A$, is the so-called +**liquid drop model**. The ansatz is based on the following expression + +$$ +BE(N,Z) = a_1A-a_2A^{2/3}-a_3\frac{Z^2}{A^{1/3}}-a_4\frac{(N-Z)^2}{A}, +$$ + +where $A$ stands for the number of nucleons and the $a_i$s are parameters which are determined by a fit +to the experimental data. + + + + +To arrive at the above expression we have assumed that we can make the following assumptions: + + * There is a volume term $a_1A$ proportional with the number of nucleons (the energy is also an extensive quantity). When an assembly of nucleons of the same size is packed together into the smallest volume, each interior nucleon has a certain number of other nucleons in contact with it. This contribution is proportional to the volume. + + * There is a surface energy term $a_2A^{2/3}$. The assumption here is that a nucleon at the surface of a nucleus interacts with fewer other nucleons than one in the interior of the nucleus and hence its binding energy is less. This surface energy term takes that into account and is therefore negative and is proportional to the surface area. + + * There is a Coulomb energy term $a_3\frac{Z^2}{A^{1/3}}$. The electric repulsion between each pair of protons in a nucleus yields less binding. + + * There is an asymmetry term $a_4\frac{(N-Z)^2}{A}$. This term is associated with the Pauli exclusion principle and reflects the fact that the proton-neutron interaction is more attractive on the average than the neutron-neutron and proton-proton interactions. + +We could also add a so-called pairing term, which is a correction term that +arises from the tendency of proton pairs and neutron pairs to +occur. An even number of particles is more stable than an odd number. + + +### Organizing our data + +Let us start with reading and organizing our data. +We start with the compilation of masses and binding energies from 2016. +After having downloaded this file to our own computer, we are now ready to read the file and start structuring our data. + + +We start with preparing folders for storing our calculations and the data file over masses and binding energies. We import also various modules that we will find useful in order to present various Machine Learning methods. Here we focus mainly on the functionality of **scikit-learn**. + +# Common imports +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +import sklearn.linear_model as skl +from sklearn.model_selection import train_test_split +from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error +import os + +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("MassEval2016.dat"),'r') + +Before we proceed, we define also a function for making our plots. You can obviously avoid this and simply set up various **matplotlib** commands every time you need them. You may however find it convenient to collect all such commands in one function and simply call this function. + +from pylab import plt, mpl +plt.style.use('seaborn') +mpl.rcParams['font.family'] = 'serif' + +def MakePlot(x,y, styles, labels, axlabels): + plt.figure(figsize=(10,6)) + for i in range(len(x)): + plt.plot(x[i], y[i], styles[i], label = labels[i]) + plt.xlabel(axlabels[0]) + plt.ylabel(axlabels[1]) + plt.legend(loc=0) + +Our next step is to read the data on experimental binding energies and +reorganize them as functions of the mass number $A$, the number of +protons $Z$ and neutrons $N$ using **pandas**. Before we do this it is +always useful (unless you have a binary file or other types of compressed +data) to actually open the file and simply take a look at it! + + +In particular, the program that outputs the final nuclear masses is written in Fortran with a specific format. It means that we need to figure out the format and which columns contain the data we are interested in. Pandas comes with a function that reads formatted output. After having admired the file, we are now ready to start massaging it with **pandas**. The file begins with some basic format information. + +""" +This is taken from the data file of the mass 2016 evaluation. +All files are 3436 lines long with 124 character per line. + Headers are 39 lines long. + col 1 : Fortran character control: 1 = page feed 0 = line feed + format : a1,i3,i5,i5,i5,1x,a3,a4,1x,f13.5,f11.5,f11.3,f9.3,1x,a2,f11.3,f9.3,1x,i3,1x,f12.5,f11.5 + These formats are reflected in the pandas widths variable below, see the statement + widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1), + Pandas has also a variable header, with length 39 in this case. +""" + +The data we are interested in are in columns 2, 3, 4 and 11, giving us +the number of neutrons, protons, mass numbers and binding energies, +respectively. We add also for the sake of completeness the element name. The data are in fixed-width formatted lines and we will +covert them into the **pandas** DataFrame structure. + +# Read the experimental data with Pandas +Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11), + names=('N', 'Z', 'A', 'Element', 'Ebinding'), + widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1), + header=39, + index_col=False) + +# Extrapolated values are indicated by '#' in place of the decimal place, so +# the Ebinding column won't be numeric. Coerce to float and drop these entries. +Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce') +Masses = Masses.dropna() +# Convert from keV to MeV. +Masses['Ebinding'] /= 1000 + +# Group the DataFrame by nucleon number, A. +Masses = Masses.groupby('A') +# Find the rows of the grouped DataFrame with the maximum binding energy. +Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()]) + +We have now read in the data, grouped them according to the variables we are interested in. +We see how easy it is to reorganize the data using **pandas**. If we +were to do these operations in C/C++ or Fortran, we would have had to +write various functions/subroutines which perform the above +reorganizations for us. Having reorganized the data, we can now start +to make some simple fits using both the functionalities in **numpy** and +**Scikit-Learn** afterwards. + +Now we define five variables which contain +the number of nucleons $A$, the number of protons $Z$ and the number of neutrons $N$, the element name and finally the energies themselves. + +A = Masses['A'] +Z = Masses['Z'] +N = Masses['N'] +Element = Masses['Element'] +Energies = Masses['Ebinding'] +print(Masses) + +The next step, and we will define this mathematically later, is to set up the so-called **design matrix**. We will throughout call this matrix $\boldsymbol{X}$. +It has dimensionality $p\times n$, where $n$ is the number of data points and $p$ are the so-called predictors. In our case here they are given by the number of polynomials in $A$ we wish to include in the fit. + +# Now we set up the design matrix X +X = np.zeros((len(A),5)) +X[:,0] = 1 +X[:,1] = A +X[:,2] = A**(2.0/3.0) +X[:,3] = A**(-1.0/3.0) +X[:,4] = A**(-1.0) + +With **scikitlearn** we are now ready to use linear regression and fit our data. + +clf = skl.LinearRegression().fit(X, Energies) +fity = clf.predict(X) + +Pretty simple! +Now we can print measures of how our fit is doing, the coefficients from the fits and plot the final fit together with our data. + +# The mean squared error +print("Mean squared error: %.2f" % mean_squared_error(Energies, fity)) +# Explained variance score: 1 is perfect prediction +print('Variance score: %.2f' % r2_score(Energies, fity)) +# Mean absolute error +print('Mean absolute error: %.2f' % mean_absolute_error(Energies, fity)) +print(clf.coef_, clf.intercept_) + +Masses['Eapprox'] = fity +# Generate a plot comparing the experimental with the fitted values values. +fig, ax = plt.subplots() +ax.set_xlabel(r'$A = N + Z$') +ax.set_ylabel(r'$E_\mathrm{bind}\,/\mathrm{MeV}$') +ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2, + label='Ame2016') +ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m', + label='Fit') +ax.legend() +save_fig("Masses2016") +plt.show() + +As a teaser, let us now see how we can do this with decision trees using **scikit-learn**. Later we will switch to so-called **random forests**! + + +#Decision Tree Regression +from sklearn.tree import DecisionTreeRegressor +regr_1=DecisionTreeRegressor(max_depth=5) +regr_2=DecisionTreeRegressor(max_depth=7) +regr_3=DecisionTreeRegressor(max_depth=9) +regr_1.fit(X, Energies) +regr_2.fit(X, Energies) +regr_3.fit(X, Energies) + + +y_1 = regr_1.predict(X) +y_2 = regr_2.predict(X) +y_3=regr_3.predict(X) +Masses['Eapprox'] = y_3 +# Plot the results +plt.figure() +plt.plot(A, Energies, color="blue", label="Data", linewidth=2) +plt.plot(A, y_1, color="red", label="max_depth=5", linewidth=2) +plt.plot(A, y_2, color="green", label="max_depth=7", linewidth=2) +plt.plot(A, y_3, color="m", label="max_depth=9", linewidth=2) + +plt.xlabel("$A$") +plt.ylabel("$E$[MeV]") +plt.title("Decision Tree Regression") +plt.legend() +save_fig("Masses2016Trees") +plt.show() +print(Masses) +print(np.mean( (Energies-y_1)**2)) + +The **seaborn** package allows us to visualize data in an efficient way. Note that we use **scikit-learn**'s multi-layer perceptron (or feed forward neural network) +functionality. + +from sklearn.neural_network import MLPRegressor +from sklearn.metrics import accuracy_score +import seaborn as sns + +X_train = X +Y_train = Energies +n_hidden_neurons = 100 +epochs = 100 +# store models for later use +eta_vals = np.logspace(-5, 1, 7) +lmbd_vals = np.logspace(-5, 1, 7) +# store the models for later use +DNN_scikit = np.zeros((len(eta_vals), len(lmbd_vals)), dtype=object) +train_accuracy = np.zeros((len(eta_vals), len(lmbd_vals))) +sns.set() +for i, eta in enumerate(eta_vals): + for j, lmbd in enumerate(lmbd_vals): + dnn = MLPRegressor(hidden_layer_sizes=(n_hidden_neurons), activation='logistic', + alpha=lmbd, learning_rate_init=eta, max_iter=epochs) + dnn.fit(X_train, Y_train) + DNN_scikit[i][j] = dnn + train_accuracy[i][j] = dnn.score(X_train, Y_train) + +fig, ax = plt.subplots(figsize = (10, 10)) +sns.heatmap(train_accuracy, annot=True, ax=ax, cmap="viridis") +ax.set_title("Training Accuracy") +ax.set_ylabel("$\eta$") +ax.set_xlabel("$\lambda$") +plt.show() + +## Linear Regression, basic elements + + +[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureAug27.mp4?vrtx=view-as-webpage). + + +Fitting a continuous function with linear parameterization in terms of the parameters $\boldsymbol{\beta}$. +* Method of choice for fitting a continuous function! + +* Gives an excellent introduction to central Machine Learning features with **understandable pedagogical** links to other methods like **Neural Networks**, **Support Vector Machines** etc + +* Analytical expression for the fitting parameters $\boldsymbol{\beta}$ + +* Analytical expressions for statistical propertiers like mean values, variances, confidence intervals and more + +* Analytical relation with probabilistic interpretations + +* Easy to introduce basic concepts like bias-variance tradeoff, cross-validation, resampling and regularization techniques and many other ML topics + +* Easy to code! And links well with classification problems and logistic regression and neural networks + +* Allows for **easy** hands-on understanding of gradient descent methods + +* and many more features + +For more discussions of Ridge and Lasso regression, [Wessel van Wieringen's](https://arxiv.org/abs/1509.09169) article is highly recommended. +Similarly, [Mehta et al's article](https://arxiv.org/abs/1803.08823) is also recommended. + + + +Regression modeling deals with the description of the sampling distribution of a given random variable $y$ and how it varies as function of another variable or a set of such variables $\boldsymbol{x} =[x_0, x_1,\dots, x_{n-1}]^T$. +The first variable is called the **dependent**, the **outcome** or the **response** variable while the set of variables $\boldsymbol{x}$ is called the independent variable, or the predictor variable or the explanatory variable. + +A regression model aims at finding a likelihood function $p(\boldsymbol{y}\vert \boldsymbol{x})$, that is the conditional distribution for $\boldsymbol{y}$ with a given $\boldsymbol{x}$. The estimation of $p(\boldsymbol{y}\vert \boldsymbol{x})$ is made using a data set with +* $n$ cases $i = 0, 1, 2, \dots, n-1$ + +* Response (target, dependent or outcome) variable $y_i$ with $i = 0, 1, 2, \dots, n-1$ + +* $p$ so-called explanatory (independent or predictor) variables $\boldsymbol{x}_i=[x_{i0}, x_{i1}, \dots, x_{ip-1}]$ with $i = 0, 1, 2, \dots, n-1$ and explanatory variables running from $0$ to $p-1$. See below for more explicit examples. + + The goal of the regression analysis is to extract/exploit relationship between $\boldsymbol{y}$ and $\boldsymbol{x}$ in or to infer causal dependencies, approximations to the likelihood functions, functional relationships and to make predictions, making fits and many other things. + + +Consider an experiment in which $p$ characteristics of $n$ samples are +measured. The data from this experiment, for various explanatory variables $p$ are normally represented by a matrix +$\mathbf{X}$. + +The matrix $\mathbf{X}$ is called the *design +matrix*. Additional information of the samples is available in the +form of $\boldsymbol{y}$ (also as above). The variable $\boldsymbol{y}$ is +generally referred to as the *response variable*. The aim of +regression analysis is to explain $\boldsymbol{y}$ in terms of +$\boldsymbol{X}$ through a functional relationship like $y_i = +f(\mathbf{X}_{i,\ast})$. When no prior knowledge on the form of +$f(\cdot)$ is available, it is common to assume a linear relationship +between $\boldsymbol{X}$ and $\boldsymbol{y}$. This assumption gives rise to +the *linear regression model* where $\boldsymbol{\beta} = [\beta_0, \ldots, +\beta_{p-1}]^{T}$ are the *regression parameters*. + +Linear regression gives us a set of analytical equations for the parameters $\beta_j$. + + +In order to understand the relation among the predictors $p$, the set of data $n$ and the target (outcome, output etc) $\boldsymbol{y}$, +consider the model we discussed for describing nuclear binding energies. + +There we assumed that we could parametrize the data using a polynomial approximation based on the liquid drop model. +Assuming + +$$ +BE(A) = a_0+a_1A+a_2A^{2/3}+a_3A^{-1/3}+a_4A^{-1}, +$$ + +we have five predictors, that is the intercept, the $A$ dependent term, the $A^{2/3}$ term and the $A^{-1/3}$ and $A^{-1}$ terms. +This gives $p=0,1,2,3,4$. Furthermore we have $n$ entries for each predictor. It means that our design matrix is a +$p\times n$ matrix $\boldsymbol{X}$. + +Here the predictors are based on a model we have made. A popular data set which is widely encountered in ML applications is the +so-called [credit card default data from Taiwan](https://www.sciencedirect.com/science/article/pii/S0957417407006719?via%3Dihub). The data set contains data on $n=30000$ credit card holders with predictors like gender, marital status, age, profession, education, etc. In total there are $24$ such predictors or attributes leading to a design matrix of dimensionality $24 \times 30000$. This is however a classification problem and we will come back to it when we discuss Logistic Regression. + + +Before we proceed let us study a case from linear algebra where we aim at fitting a set of data $\boldsymbol{y}=[y_0,y_1,\dots,y_{n-1}]$. We could think of these data as a result of an experiment or a complicated numerical experiment. These data are functions of a series of variables $\boldsymbol{x}=[x_0,x_1,\dots,x_{n-1}]$, that is $y_i = y(x_i)$ with $i=0,1,2,\dots,n-1$. The variables $x_i$ could represent physical quantities like time, temperature, position etc. We assume that $y(x)$ is a smooth function. + +Since obtaining these data points may not be trivial, we want to use these data to fit a function which can allow us to make predictions for values of $y$ which are not in the present set. The perhaps simplest approach is to assume we can parametrize our function in terms of a polynomial of degree $n-1$ with $n$ points, that is + +$$ +y=y(x) \rightarrow y(x_i)=\tilde{y}_i+\epsilon_i=\sum_{j=0}^{n-1} \beta_j x_i^j+\epsilon_i, +$$ + +where $\epsilon_i$ is the error in our approximation. + + +For every set of values $y_i,x_i$ we have thus the corresponding set of equations + +$$ +\begin{align*} +y_0&=\beta_0+\beta_1x_0^1+\beta_2x_0^2+\dots+\beta_{n-1}x_0^{n-1}+\epsilon_0\\ +y_1&=\beta_0+\beta_1x_1^1+\beta_2x_1^2+\dots+\beta_{n-1}x_1^{n-1}+\epsilon_1\\ +y_2&=\beta_0+\beta_1x_2^1+\beta_2x_2^2+\dots+\beta_{n-1}x_2^{n-1}+\epsilon_2\\ +\dots & \dots \\ +y_{n-1}&=\beta_0+\beta_1x_{n-1}^1+\beta_2x_{n-1}^2+\dots+\beta_{n-1}x_{n-1}^{n-1}+\epsilon_{n-1}.\\ +\end{align*} +$$ + +Defining the vectors + +$$ +\boldsymbol{y} = [y_0,y_1, y_2,\dots, y_{n-1}]^T, +$$ + +and + +$$ +\boldsymbol{\beta} = [\beta_0,\beta_1, \beta_2,\dots, \beta_{n-1}]^T, +$$ + +and + +$$ +\boldsymbol{\epsilon} = [\epsilon_0,\epsilon_1, \epsilon_2,\dots, \epsilon_{n-1}]^T, +$$ + +and the design matrix + +$$ +\boldsymbol{X}= +\begin{bmatrix} +1& x_{0}^1 &x_{0}^2& \dots & \dots &x_{0}^{n-1}\\ +1& x_{1}^1 &x_{1}^2& \dots & \dots &x_{1}^{n-1}\\ +1& x_{2}^1 &x_{2}^2& \dots & \dots &x_{2}^{n-1}\\ +\dots& \dots &\dots& \dots & \dots &\dots\\ +1& x_{n-1}^1 &x_{n-1}^2& \dots & \dots &x_{n-1}^{n-1}\\ +\end{bmatrix} +$$ + +we can rewrite our equations as + +$$ +\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta}+\boldsymbol{\epsilon}. +$$ + +The above design matrix is called a [Vandermonde matrix](https://en.wikipedia.org/wiki/Vandermonde_matrix). + +We are obviously not limited to the above polynomial expansions. We +could replace the various powers of $x$ with elements of Fourier +series or instead of $x_i^j$ we could have $\cos{(j x_i)}$ or $\sin{(j +x_i)}$, or time series or other orthogonal functions. For every set +of values $y_i,x_i$ we can then generalize the equations to + +$$ +\begin{align*} +y_0&=\beta_0x_{00}+\beta_1x_{01}+\beta_2x_{02}+\dots+\beta_{n-1}x_{0n-1}+\epsilon_0\\ +y_1&=\beta_0x_{10}+\beta_1x_{11}+\beta_2x_{12}+\dots+\beta_{n-1}x_{1n-1}+\epsilon_1\\ +y_2&=\beta_0x_{20}+\beta_1x_{21}+\beta_2x_{22}+\dots+\beta_{n-1}x_{2n-1}+\epsilon_2\\ +\dots & \dots \\ +y_{i}&=\beta_0x_{i0}+\beta_1x_{i1}+\beta_2x_{i2}+\dots+\beta_{n-1}x_{in-1}+\epsilon_i\\ +\dots & \dots \\ +y_{n-1}&=\beta_0x_{n-1,0}+\beta_1x_{n-1,2}+\beta_2x_{n-1,2}+\dots+\beta_{n-1}x_{n-1,n-1}+\epsilon_{n-1}.\\ +\end{align*} +$$ + +**Note that we have $p=n$ here. The matrix is symmetric. This is generally not the case!** + +We redefine in turn the matrix $\boldsymbol{X}$ as + +$$ +\boldsymbol{X}= +\begin{bmatrix} +x_{00}& x_{01} &x_{02}& \dots & \dots &x_{0,n-1}\\ +x_{10}& x_{11} &x_{12}& \dots & \dots &x_{1,n-1}\\ +x_{20}& x_{21} &x_{22}& \dots & \dots &x_{2,n-1}\\ +\dots& \dots &\dots& \dots & \dots &\dots\\ +x_{n-1,0}& x_{n-1,1} &x_{n-1,2}& \dots & \dots &x_{n-1,n-1}\\ +\end{bmatrix} +$$ + +and without loss of generality we rewrite again our equations as + +$$ +\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta}+\boldsymbol{\epsilon}. +$$ + +The left-hand side of this equation is kwown. Our error vector $\boldsymbol{\epsilon}$ and the parameter vector $\boldsymbol{\beta}$ are our unknow quantities. How can we obtain the optimal set of $\beta_i$ values? + +We have defined the matrix $\boldsymbol{X}$ via the equations + +$$ +\begin{align*} +y_0&=\beta_0x_{00}+\beta_1x_{01}+\beta_2x_{02}+\dots+\beta_{n-1}x_{0n-1}+\epsilon_0\\ +y_1&=\beta_0x_{10}+\beta_1x_{11}+\beta_2x_{12}+\dots+\beta_{n-1}x_{1n-1}+\epsilon_1\\ +y_2&=\beta_0x_{20}+\beta_1x_{21}+\beta_2x_{22}+\dots+\beta_{n-1}x_{2n-1}+\epsilon_1\\ +\dots & \dots \\ +y_{i}&=\beta_0x_{i0}+\beta_1x_{i1}+\beta_2x_{i2}+\dots+\beta_{n-1}x_{in-1}+\epsilon_1\\ +\dots & \dots \\ +y_{n-1}&=\beta_0x_{n-1,0}+\beta_1x_{n-1,2}+\beta_2x_{n-1,2}+\dots+\beta_{n-1}x_{n-1,n-1}+\epsilon_{n-1}.\\ +\end{align*} +$$ + +As we noted above, we stayed with a system with the design matrix + $\boldsymbol{X}\in {\mathbb{R}}^{n\times n}$, that is we have $p=n$. For reasons to come later (algorithmic arguments) we will hereafter define +our matrix as $\boldsymbol{X}\in {\mathbb{R}}^{n\times p}$, with the predictors refering to the column numbers and the entries $n$ being the row elements. + +In our [introductory notes](https://compphysics.github.io/MachineLearning/doc/pub/How2ReadData/html/How2ReadData.html) we looked at the so-called [liquid drop model](https://en.wikipedia.org/wiki/Semi-empirical_mass_formula). Let us remind ourselves about what we did by looking at the code. + +We restate the parts of the code we are most interested in. + +# Common imports +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +from IPython.display import display +import os + +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("MassEval2016.dat"),'r') + + +# Read the experimental data with Pandas +Masses = pd.read_fwf(infile, usecols=(2,3,4,6,11), + names=('N', 'Z', 'A', 'Element', 'Ebinding'), + widths=(1,3,5,5,5,1,3,4,1,13,11,11,9,1,2,11,9,1,3,1,12,11,1), + header=39, + index_col=False) + +# Extrapolated values are indicated by '#' in place of the decimal place, so +# the Ebinding column won't be numeric. Coerce to float and drop these entries. +Masses['Ebinding'] = pd.to_numeric(Masses['Ebinding'], errors='coerce') +Masses = Masses.dropna() +# Convert from keV to MeV. +Masses['Ebinding'] /= 1000 + +# Group the DataFrame by nucleon number, A. +Masses = Masses.groupby('A') +# Find the rows of the grouped DataFrame with the maximum binding energy. +Masses = Masses.apply(lambda t: t[t.Ebinding==t.Ebinding.max()]) +A = Masses['A'] +Z = Masses['Z'] +N = Masses['N'] +Element = Masses['Element'] +Energies = Masses['Ebinding'] + +# Now we set up the design matrix X +X = np.zeros((len(A),5)) +X[:,0] = 1 +X[:,1] = A +X[:,2] = A**(2.0/3.0) +X[:,3] = A**(-1.0/3.0) +X[:,4] = A**(-1.0) +# Then nice printout using pandas +DesignMatrix = pd.DataFrame(X) +DesignMatrix.index = A +DesignMatrix.columns = ['1', 'A', 'A^(2/3)', 'A^(-1/3)', '1/A'] +display(DesignMatrix) + +With $\boldsymbol{\beta}\in {\mathbb{R}}^{p\times 1}$, it means that we will hereafter write our equations for the approximation as + +$$ +\boldsymbol{\tilde{y}}= \boldsymbol{X}\boldsymbol{\beta}, +$$ + +throughout these lectures. + +With the above we use the design matrix to define the approximation $\boldsymbol{\tilde{y}}$ via the unknown quantity $\boldsymbol{\beta}$ as + +$$ +\boldsymbol{\tilde{y}}= \boldsymbol{X}\boldsymbol{\beta}, +$$ + +and in order to find the optimal parameters $\beta_i$ instead of solving the above linear algebra problem, we define a function which gives a measure of the spread between the values $y_i$ (which represent hopefully the exact values) and the parameterized values $\tilde{y}_i$, namely + +$$ +C(\boldsymbol{\beta})=\frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)^T\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)\right\}, +$$ + +or using the matrix $\boldsymbol{X}$ and in a more compact matrix-vector notation as + +$$ +C(\boldsymbol{\beta})=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. +$$ + +This function is one possible way to define the so-called cost function. + + + +It is also common to define +the function $C$ as + +$$ +C(\boldsymbol{\beta})=\frac{1}{2n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2, +$$ + +since when taking the first derivative with respect to the unknown parameters $\beta$, the factor of $2$ cancels out. + +The function + +$$ +C(\boldsymbol{\beta})=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}, +$$ + +can be linked to the variance of the quantity $y_i$ if we interpret the latter as the mean value. +When linking (see the discussion below) with the maximum likelihood approach below, we will indeed interpret $y_i$ as a mean value + +$$ +y_{i}=\langle y_i \rangle = \beta_0x_{i,0}+\beta_1x_{i,1}+\beta_2x_{i,2}+\dots+\beta_{n-1}x_{i,n-1}+\epsilon_i, +$$ + +where $\langle y_i \rangle$ is the mean value. Keep in mind also that +till now we have treated $y_i$ as the exact value. Normally, the +response (dependent or outcome) variable $y_i$ the outcome of a +numerical experiment or another type of experiment and is thus only an +approximation to the true value. It is then always accompanied by an +error estimate, often limited to a statistical error estimate given by +the standard deviation discussed earlier. In the discussion here we +will treat $y_i$ as our exact value for the response variable. + +In order to find the parameters $\beta_i$ we will then minimize the spread of $C(\boldsymbol{\beta})$, that is we are going to solve the problem + +$$ +{\displaystyle \min_{\boldsymbol{\beta}\in +{\mathbb{R}}^{p}}}\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. +$$ + +In practical terms it means we will require + +$$ +\frac{\partial C(\boldsymbol{\beta})}{\partial \beta_j} = \frac{\partial }{\partial \beta_j}\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}\right)^2\right]=0, +$$ + +which results in + +$$ +\frac{\partial C(\boldsymbol{\beta})}{\partial \beta_j} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}x_{ij}\left(y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}\right)\right]=0, +$$ + +or in a matrix-vector form as + +$$ +\frac{\partial C(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right). +$$ + +We can rewrite + +$$ +\frac{\partial C(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right), +$$ + +as + +$$ +\boldsymbol{X}^T\boldsymbol{y} = \boldsymbol{X}^T\boldsymbol{X}\boldsymbol{\beta}, +$$ + +and if the matrix $\boldsymbol{X}^T\boldsymbol{X}$ is invertible we have the solution + +$$ +\boldsymbol{\beta} =\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}. +$$ + +We note also that since our design matrix is defined as $\boldsymbol{X}\in +{\mathbb{R}}^{n\times p}$, the product $\boldsymbol{X}^T\boldsymbol{X} \in +{\mathbb{R}}^{p\times p}$. In the above case we have that $p \ll n$, +in our case $p=5$ meaning that we end up with inverting a small +$5\times 5$ matrix. This is a rather common situation, in many cases we end up with low-dimensional +matrices to invert. The methods discussed here and for many other +supervised learning algorithms like classification with logistic +regression or support vector machines, exhibit dimensionalities which +allow for the usage of direct linear algebra methods such as **LU** decomposition or **Singular Value Decomposition** (SVD) for finding the inverse of the matrix +$\boldsymbol{X}^T\boldsymbol{X}$. + +**Small question**: Do you think the example we have at hand here (the nuclear binding energies) can lead to problems in inverting the matrix $\boldsymbol{X}^T\boldsymbol{X}$? What kind of problems can we expect? + + +The following matrix and vector relation will be useful here and for the rest of the course. Vectors are always written as boldfaced lower case letters and +matrices as upper case boldfaced letters. + +4 +8 + +< +< +< +! +! +M +A +T +H +_ +B +L +O +C +K + +4 +9 + +< +< +< +! +! +M +A +T +H +_ +B +L +O +C +K + +5 +0 + +< +< +< +! +! +M +A +T +H +_ +B +L +O +C +K + +$$ +\frac{\partial \log{\vert\boldsymbol{A}\vert}}{\partial \boldsymbol{A}} = (\boldsymbol{A}^{-1})^T. +$$ + +The residuals $\boldsymbol{\epsilon}$ are in turn given by + +$$ +\boldsymbol{\epsilon} = \boldsymbol{y}-\boldsymbol{\tilde{y}} = \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}, +$$ + +and with + +$$ +\boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)= 0, +$$ + +we have + +$$ +\boldsymbol{X}^T\boldsymbol{\epsilon}=\boldsymbol{X}^T\left( \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)= 0, +$$ + +meaning that the solution for $\boldsymbol{\beta}$ is the one which minimizes the residuals. Later we will link this with the maximum likelihood approach. + + +Let us now return to our nuclear binding energies and simply code the above equations. + + +It is rather straightforward to implement the matrix inversion and obtain the parameters $\boldsymbol{\beta}$. After having defined the matrix $\boldsymbol{X}$ we simply need to +write + +# matrix inversion to find beta +beta = np.linalg.inv(X.T.dot(X)).dot(X.T).dot(Energies) +# and then make the prediction +ytilde = X @ beta + +Alternatively, you can use the least squares functionality in **Numpy** as + +fit = np.linalg.lstsq(X, Energies, rcond =None)[0] +ytildenp = np.dot(fit,X.T) + +And finally we plot our fit with and compare with data + +Masses['Eapprox'] = ytilde +# Generate a plot comparing the experimental with the fitted values values. +fig, ax = plt.subplots() +ax.set_xlabel(r'$A = N + Z$') +ax.set_ylabel(r'$E_\mathrm{bind}\,/\mathrm{MeV}$') +ax.plot(Masses['A'], Masses['Ebinding'], alpha=0.7, lw=2, + label='Ame2016') +ax.plot(Masses['A'], Masses['Eapprox'], alpha=0.7, lw=2, c='m', + label='Fit') +ax.legend() +save_fig("Masses2016OLS") +plt.show() + +We can easily test our fit by computing the $R2$ score that we discussed in connection with the functionality of **Scikit-Learn** in the introductory slides. +Since we are not using **Scikit-Learn** here we can define our own $R2$ function as + +def R2(y_data, y_model): + return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2) + +and we would be using it as + +print(R2(Energies,ytilde)) + +We can easily add our **MSE** score as + +def MSE(y_data,y_model): + n = np.size(y_model) + return np.sum((y_data-y_model)**2)/n + +print(MSE(Energies,ytilde)) + +and finally the relative error as + +def RelativeError(y_data,y_model): + return abs((y_data-y_model)/y_data) +print(RelativeError(Energies, ytilde)) + +### The $\chi^2$ function + +Normally, the response (dependent or outcome) variable $y_i$ is the +outcome of a numerical experiment or another type of experiment and is +thus only an approximation to the true value. It is then always +accompanied by an error estimate, often limited to a statistical error +estimate given by the standard deviation discussed earlier. In the +discussion here we will treat $y_i$ as our exact value for the +response variable. + +Introducing the standard deviation $\sigma_i$ for each measurement +$y_i$, we define now the $\chi^2$ function (omitting the $1/n$ term) +as + +$$ +\chi^2(\boldsymbol{\beta})=\frac{1}{n}\sum_{i=0}^{n-1}\frac{\left(y_i-\tilde{y}_i\right)^2}{\sigma_i^2}=\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)^T\frac{1}{\boldsymbol{\Sigma^2}}\left(\boldsymbol{y}-\boldsymbol{\tilde{y}}\right)\right\}, +$$ + +where the matrix $\boldsymbol{\Sigma}$ is a diagonal matrix with $\sigma_i$ as matrix elements. + + +In order to find the parameters $\beta_i$ we will then minimize the spread of $\chi^2(\boldsymbol{\beta})$ by requiring + +$$ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_j} = \frac{\partial }{\partial \beta_j}\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(\frac{y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}}{\sigma_i}\right)^2\right]=0, +$$ + +which results in + +$$ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_j} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}\frac{x_{ij}}{\sigma_i}\left(\frac{y_i-\beta_0x_{i,0}-\beta_1x_{i,1}-\beta_2x_{i,2}-\dots-\beta_{n-1}x_{i,n-1}}{\sigma_i}\right)\right]=0, +$$ + +or in a matrix-vector form as + +$$ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{A}^T\left( \boldsymbol{b}-\boldsymbol{A}\boldsymbol{\beta}\right). +$$ + +where we have defined the matrix $\boldsymbol{A} =\boldsymbol{X}/\boldsymbol{\Sigma}$ with matrix elements $a_{ij} = x_{ij}/\sigma_i$ and the vector $\boldsymbol{b}$ with elements $b_i = y_i/\sigma_i$. + +We can rewrite + +$$ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = 0 = \boldsymbol{A}^T\left( \boldsymbol{b}-\boldsymbol{A}\boldsymbol{\beta}\right), +$$ + +as + +$$ +\boldsymbol{A}^T\boldsymbol{b} = \boldsymbol{A}^T\boldsymbol{A}\boldsymbol{\beta}, +$$ + +and if the matrix $\boldsymbol{A}^T\boldsymbol{A}$ is invertible we have the solution + +$$ +\boldsymbol{\beta} =\left(\boldsymbol{A}^T\boldsymbol{A}\right)^{-1}\boldsymbol{A}^T\boldsymbol{b}. +$$ + +If we then introduce the matrix + +$$ +\boldsymbol{H} = \left(\boldsymbol{A}^T\boldsymbol{A}\right)^{-1}, +$$ + +we have then the following expression for the parameters $\beta_j$ (the matrix elements of $\boldsymbol{H}$ are $h_{ij}$) + +$$ +\beta_j = \sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}\frac{y_i}{\sigma_i}\frac{x_{ik}}{\sigma_i} = \sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}b_ia_{ik} +$$ + +We state without proof the expression for the uncertainty in the parameters $\beta_j$ as (we leave this as an exercise) + +$$ +\sigma^2(\beta_j) = \sum_{i=0}^{n-1}\sigma_i^2\left( \frac{\partial \beta_j}{\partial y_i}\right)^2, +$$ + +resulting in + +$$ +\sigma^2(\beta_j) = \left(\sum_{k=0}^{p-1}h_{jk}\sum_{i=0}^{n-1}a_{ik}\right)\left(\sum_{l=0}^{p-1}h_{jl}\sum_{m=0}^{n-1}a_{ml}\right) = h_{jj}! +$$ + +The first step here is to approximate the function $y$ with a first-order polynomial, that is we write + +$$ +y=y(x) \rightarrow y(x_i) \approx \beta_0+\beta_1 x_i. +$$ + +By computing the derivatives of $\chi^2$ with respect to $\beta_0$ and $\beta_1$ show that these are given by + +$$ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_0} = -2\left[ \frac{1}{n}\sum_{i=0}^{n-1}\left(\frac{y_i-\beta_0-\beta_1x_{i}}{\sigma_i^2}\right)\right]=0, +$$ + +and + +$$ +\frac{\partial \chi^2(\boldsymbol{\beta})}{\partial \beta_1} = -\frac{2}{n}\left[ \sum_{i=0}^{n-1}x_i\left(\frac{y_i-\beta_0-\beta_1x_{i}}{\sigma_i^2}\right)\right]=0. +$$ + +For a linear fit (a first-order polynomial) we don't need to invert a matrix!! +Defining + +$$ +\gamma = \sum_{i=0}^{n-1}\frac{1}{\sigma_i^2}, +$$ + +$$ +\gamma_x = \sum_{i=0}^{n-1}\frac{x_{i}}{\sigma_i^2}, +$$ + +$$ +\gamma_y = \sum_{i=0}^{n-1}\left(\frac{y_i}{\sigma_i^2}\right), +$$ + +$$ +\gamma_{xx} = \sum_{i=0}^{n-1}\frac{x_ix_{i}}{\sigma_i^2}, +$$ + +$$ +\gamma_{xy} = \sum_{i=0}^{n-1}\frac{y_ix_{i}}{\sigma_i^2}, +$$ + +we obtain + +$$ +\beta_0 = \frac{\gamma_{xx}\gamma_y-\gamma_x\gamma_y}{\gamma\gamma_{xx}-\gamma_x^2}, +$$ + +$$ +\beta_1 = \frac{\gamma_{xy}\gamma-\gamma_x\gamma_y}{\gamma\gamma_{xx}-\gamma_x^2}. +$$ + +This approach (different linear and non-linear regression) suffers +often from both being underdetermined and overdetermined in the +unknown coefficients $\beta_i$. A better approach is to use the +Singular Value Decomposition (SVD) method discussed below. Or using +Lasso and Ridge regression. See below. + + +### Fitting an Equation of State for Dense Nuclear Matter + +Before we continue, let us introduce yet another example. We are going to fit the +nuclear equation of state using results from many-body calculations. +The equation of state we have made available here, as function of +density, has been derived using modern nucleon-nucleon potentials with +[the addition of three-body +forces](https://www.sciencedirect.com/science/article/pii/S0370157399001106). This +time the file is presented as a standard **csv** file. + +The beginning of the Python code here is similar to what you have seen +before, with the same initializations and declarations. We use also +**pandas** again, rather extensively in order to organize our data. + +The difference now is that we use **Scikit-Learn's** regression tools +instead of our own matrix inversion implementation. Furthermore, we +sneak in **Ridge** regression (to be discussed below) which includes a +hyperparameter $\lambda$, also to be explained below. + +# Common imports +import os +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +import matplotlib.pyplot as plt +import sklearn.linear_model as skl +from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error + +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("EoS.csv"),'r') + +# Read the EoS data as csv file and organize the data into two arrays with density and energies +EoS = pd.read_csv(infile, names=('Density', 'Energy')) +EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce') +EoS = EoS.dropna() +Energies = EoS['Energy'] +Density = EoS['Density'] +# The design matrix now as function of various polytrops +X = np.zeros((len(Density),4)) +X[:,3] = Density**(4.0/3.0) +X[:,2] = Density +X[:,1] = Density**(2.0/3.0) +X[:,0] = 1 + +# We use now Scikit-Learn's linear regressor and ridge regressor +# OLS part +clf = skl.LinearRegression().fit(X, Energies) +ytilde = clf.predict(X) +EoS['Eols'] = ytilde +# The mean squared error +print("Mean squared error: %.2f" % mean_squared_error(Energies, ytilde)) +# Explained variance score: 1 is perfect prediction +print('Variance score: %.2f' % r2_score(Energies, ytilde)) +# Mean absolute error +print('Mean absolute error: %.2f' % mean_absolute_error(Energies, ytilde)) +print(clf.coef_, clf.intercept_) + +# The Ridge regression with a hyperparameter lambda = 0.1 +_lambda = 0.1 +clf_ridge = skl.Ridge(alpha=_lambda).fit(X, Energies) +yridge = clf_ridge.predict(X) +EoS['Eridge'] = yridge +# The mean squared error +print("Mean squared error: %.2f" % mean_squared_error(Energies, yridge)) +# Explained variance score: 1 is perfect prediction +print('Variance score: %.2f' % r2_score(Energies, yridge)) +# Mean absolute error +print('Mean absolute error: %.2f' % mean_absolute_error(Energies, yridge)) +print(clf_ridge.coef_, clf_ridge.intercept_) + +fig, ax = plt.subplots() +ax.set_xlabel(r'$\rho[\mathrm{fm}^{-3}]$') +ax.set_ylabel(r'Energy per particle') +ax.plot(EoS['Density'], EoS['Energy'], alpha=0.7, lw=2, + label='Theoretical data') +ax.plot(EoS['Density'], EoS['Eols'], alpha=0.7, lw=2, c='m', + label='OLS') +ax.plot(EoS['Density'], EoS['Eridge'], alpha=0.7, lw=2, c='g', + label='Ridge $\lambda = 0.1$') +ax.legend() +save_fig("EoSfitting") +plt.show() + +The above simple polynomial in density $\rho$ gives an excellent fit +to the data. + +We note also that there is a small deviation between the +standard OLS and the Ridge regression at higher densities. We discuss this in more detail +below. + + +## Splitting our Data in Training and Test data + +It is normal in essentially all Machine Learning studies to split the +data in a training set and a test set (sometimes also an additional +validation set). **Scikit-Learn** has an own function for this. There +is no explicit recipe for how much data should be included as training +data and say test data. An accepted rule of thumb is to use +approximately $2/3$ to $4/5$ of the data as training data. We will +postpone a discussion of this splitting to the end of these notes and +our discussion of the so-called **bias-variance** tradeoff. Here we +limit ourselves to repeat the above equation of state fitting example +but now splitting the data into a training set and a test set. + +import os +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +from sklearn.model_selection import train_test_split +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +def R2(y_data, y_model): + return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2) +def MSE(y_data,y_model): + n = np.size(y_model) + return np.sum((y_data-y_model)**2)/n + +infile = open(data_path("EoS.csv"),'r') + +# Read the EoS data as csv file and organized into two arrays with density and energies +EoS = pd.read_csv(infile, names=('Density', 'Energy')) +EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce') +EoS = EoS.dropna() +Energies = EoS['Energy'] +Density = EoS['Density'] +# The design matrix now as function of various polytrops +X = np.zeros((len(Density),5)) +X[:,0] = 1 +X[:,1] = Density**(2.0/3.0) +X[:,2] = Density +X[:,3] = Density**(4.0/3.0) +X[:,4] = Density**(5.0/3.0) +# We split the data in test and training data +X_train, X_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2) +# matrix inversion to find beta +beta = np.linalg.inv(X_train.T.dot(X_train)).dot(X_train.T).dot(y_train) +# and then make the prediction +ytilde = X_train @ beta +print("Training R2") +print(R2(y_train,ytilde)) +print("Training MSE") +print(MSE(y_train,ytilde)) +ypredict = X_test @ beta +print("Test R2") +print(R2(y_test,ypredict)) +print("Test MSE") +print(MSE(y_test,ypredict)) + +## The Boston housing data example + +The Boston housing +data set was originally a part of UCI Machine Learning Repository +and has been removed now. The data set is now included in **Scikit-Learn**'s +library. There are 506 samples and 13 feature (predictor) variables +in this data set. The objective is to predict the value of prices of +the house using the features (predictors) listed here. + +The features/predictors are +1. CRIM: Per capita crime rate by town + +2. ZN: Proportion of residential land zoned for lots over 25000 square feet + +3. INDUS: Proportion of non-retail business acres per town + +4. CHAS: Charles River dummy variable (= 1 if tract bounds river; 0 otherwise) + +5. NOX: Nitric oxide concentration (parts per 10 million) + +6. RM: Average number of rooms per dwelling + +7. AGE: Proportion of owner-occupied units built prior to 1940 + +8. DIS: Weighted distances to five Boston employment centers + +9. RAD: Index of accessibility to radial highways + +10. TAX: Full-value property tax rate per USD10000 + +11. B: $1000(Bk - 0.63)^2$, where $Bk$ is the proportion of [people of African American descent] by town + +12. LSTAT: Percentage of lower status of the population + +13. MEDV: Median value of owner-occupied homes in USD 1000s + +## Housing data, the code +We start by importing the libraries + +import numpy as np +import matplotlib.pyplot as plt + +import pandas as pd +import seaborn as sns + +and load the Boston Housing DataSet from **Scikit-Learn** + +from sklearn.datasets import load_boston + +boston_dataset = load_boston() + +# boston_dataset is a dictionary +# let's check what it contains +boston_dataset.keys() + +Then we invoke Pandas + +boston = pd.DataFrame(boston_dataset.data, columns=boston_dataset.feature_names) +boston.head() +boston['MEDV'] = boston_dataset.target + +and preprocess the data + +# check for missing values in all the columns +boston.isnull().sum() + +We can then visualize the data + +# set the size of the figure +sns.set(rc={'figure.figsize':(11.7,8.27)}) + +# plot a histogram showing the distribution of the target values +sns.distplot(boston['MEDV'], bins=30) +plt.show() + +It is now useful to look at the correlation matrix + +# compute the pair wise correlation for all columns +correlation_matrix = boston.corr().round(2) +# use the heatmap function from seaborn to plot the correlation matrix +# annot = True to print the values inside the square +sns.heatmap(data=correlation_matrix, annot=True) + +From the above coorelation plot we can see that **MEDV** is strongly correlated to **LSTAT** and **RM**. We see also that **RAD** and **TAX** are stronly correlated, but we don't include this in our features together to avoid multi-colinearity + +plt.figure(figsize=(20, 5)) + +features = ['LSTAT', 'RM'] +target = boston['MEDV'] + +for i, col in enumerate(features): + plt.subplot(1, len(features) , i+1) + x = boston[col] + y = target + plt.scatter(x, y, marker='o') + plt.title(col) + plt.xlabel(col) + plt.ylabel('MEDV') + +Now we start training our model + +X = pd.DataFrame(np.c_[boston['LSTAT'], boston['RM']], columns = ['LSTAT','RM']) +Y = boston['MEDV'] + +We split the data into training and test sets + +from sklearn.model_selection import train_test_split + +# splits the training and test data set in 80% : 20% +# assign random_state to any value.This ensures consistency. +X_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size = 0.2, random_state=5) +print(X_train.shape) +print(X_test.shape) +print(Y_train.shape) +print(Y_test.shape) + +Then we use the linear regression functionality from **Scikit-Learn** + +from sklearn.linear_model import LinearRegression +from sklearn.metrics import mean_squared_error, r2_score + +lin_model = LinearRegression() +lin_model.fit(X_train, Y_train) + +# model evaluation for training set + +y_train_predict = lin_model.predict(X_train) +rmse = (np.sqrt(mean_squared_error(Y_train, y_train_predict))) +r2 = r2_score(Y_train, y_train_predict) + +print("The model performance for training set") +print("--------------------------------------") +print('RMSE is {}'.format(rmse)) +print('R2 score is {}'.format(r2)) +print("\n") + +# model evaluation for testing set + +y_test_predict = lin_model.predict(X_test) +# root mean square error of the model +rmse = (np.sqrt(mean_squared_error(Y_test, y_test_predict))) + +# r-squared score of the model +r2 = r2_score(Y_test, y_test_predict) + +print("The model performance for testing set") +print("--------------------------------------") +print('RMSE is {}'.format(rmse)) +print('R2 score is {}'.format(r2)) + +# plotting the y_test vs y_pred +# ideally should have been a straight line +plt.scatter(Y_test, y_test_predict) +plt.show() + +## Reducing the number of degrees of freedom, overarching view + +Many Machine Learning problems involve thousands or even millions of +features for each training instance. Not only does this make training +extremely slow, it can also make it much harder to find a good +solution, as we will see. This problem is often referred to as the +curse of dimensionality. Fortunately, in real-world problems, it is +often possible to reduce the number of features considerably, turning +an intractable problem into a tractable one. + +Later we will discuss some of the most popular dimensionality reduction +techniques: the principal component analysis (PCA), Kernel PCA, and +Locally Linear Embedding (LLE). + + +Principal component analysis and its various variants deal with the +problem of fitting a low-dimensional [affine +subspace](https://en.wikipedia.org/wiki/Affine_space) to a set of of +data points in a high-dimensional space. With its family of methods it +is one of the most used tools in data modeling, compression and +visualization. + + +Before we proceed however, we will discuss how to preprocess our +data. Till now and in connection with our previous examples we have +not met so many cases where we are too sensitive to the scaling of our +data. Normally the data may need a rescaling and/or may be sensitive +to extreme values. Scaling the data renders our inputs much more +suitable for the algorithms we want to employ. + +**Scikit-Learn** has several functions which allow us to rescale the +data, normally resulting in much better results in terms of various +accuracy scores. The **StandardScaler** function in **Scikit-Learn** +ensures that for each feature/predictor we study the mean value is +zero and the variance is one (every column in the design/feature +matrix). This scaling has the drawback that it does not ensure that +we have a particular maximum or minimum in our data set. Another +function included in **Scikit-Learn** is the **MinMaxScaler** which +ensures that all features are exactly between $0$ and $1$. The + + +The **Normalizer** scales each data +point such that the feature vector has a euclidean length of one. In other words, it +projects a data point on the circle (or sphere in the case of higher dimensions) with a +radius of 1. This means every data point is scaled by a different number (by the +inverse of it’s length). +This normalization is often used when only the direction (or angle) of the data matters, +not the length of the feature vector. + +The **RobustScaler** works similarly to the StandardScaler in that it +ensures statistical properties for each feature that guarantee that +they are on the same scale. However, the RobustScaler uses the median +and quartiles, instead of mean and variance. This makes the +RobustScaler ignore data points that are very different from the rest +(like measurement errors). These odd data points are also called +outliers, and might often lead to trouble for other scaling +techniques. + + +### Simple preprocessing examples, Franke function and regression + +# Common imports +import os +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +import sklearn.linear_model as skl +from sklearn.metrics import mean_squared_error +from sklearn.model_selection import train_test_split +from sklearn.preprocessing import MinMaxScaler, StandardScaler, Normalizer + +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + + +def FrankeFunction(x,y): + term1 = 0.75*np.exp(-(0.25*(9*x-2)**2) - 0.25*((9*y-2)**2)) + term2 = 0.75*np.exp(-((9*x+1)**2)/49.0 - 0.1*(9*y+1)) + term3 = 0.5*np.exp(-(9*x-7)**2/4.0 - 0.25*((9*y-3)**2)) + term4 = -0.2*np.exp(-(9*x-4)**2 - (9*y-7)**2) + return term1 + term2 + term3 + term4 + + +def create_X(x, y, n ): + if len(x.shape) > 1: + x = np.ravel(x) + y = np.ravel(y) + + N = len(x) + l = int((n+1)*(n+2)/2) # Number of elements in beta + X = np.ones((N,l)) + + for i in range(1,n+1): + q = int((i)*(i+1)/2) + for k in range(i+1): + X[:,q+k] = (x**(i-k))*(y**k) + + return X + + +# Making meshgrid of datapoints and compute Franke's function +n = 5 +N = 1000 +x = np.sort(np.random.uniform(0, 1, N)) +y = np.sort(np.random.uniform(0, 1, N)) +z = FrankeFunction(x, y) +X = create_X(x, y, n=n) +# split in training and test data +X_train, X_test, y_train, y_test = train_test_split(X,z,test_size=0.2) + + +clf = skl.LinearRegression().fit(X_train, y_train) + +# The mean squared error and R2 score +print("MSE before scaling: {:.2f}".format(mean_squared_error(clf.predict(X_test), y_test))) +print("R2 score before scaling {:.2f}".format(clf.score(X_test,y_test))) + +scaler = StandardScaler() +scaler.fit(X_train) +X_train_scaled = scaler.transform(X_train) +X_test_scaled = scaler.transform(X_test) + +print("Feature min values before scaling:\n {}".format(X_train.min(axis=0))) +print("Feature max values before scaling:\n {}".format(X_train.max(axis=0))) + +print("Feature min values after scaling:\n {}".format(X_train_scaled.min(axis=0))) +print("Feature max values after scaling:\n {}".format(X_train_scaled.max(axis=0))) + +clf = skl.LinearRegression().fit(X_train_scaled, y_train) + + +print("MSE after scaling: {:.2f}".format(mean_squared_error(clf.predict(X_test_scaled), y_test))) +print("R2 score for scaled data: {:.2f}".format(clf.score(X_test_scaled,y_test))) \ No newline at end of file diff --git a/doc/LectureNotes/_build/jupyter_execute/chapter2.ipynb b/doc/LectureNotes/_build/jupyter_execute/chapter2.ipynb new file mode 100644 index 000000000..4086237a5 --- /dev/null +++ b/doc/LectureNotes/_build/jupyter_execute/chapter2.ipynb @@ -0,0 +1,1420 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Resampling Methods\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSept3.mp4?vrtx=view-as-webpage)\n", + "\n", + "\n", + "## Introduction\n", + "\n", + "Resampling methods are an indispensable tool in modern\n", + "statistics. They involve repeatedly drawing samples from a training\n", + "set and refitting a model of interest on each sample in order to\n", + "obtain additional information about the fitted model. For example, in\n", + "order to estimate the variability of a linear regression fit, we can\n", + "repeatedly draw different samples from the training data, fit a linear\n", + "regression to each new sample, and then examine the extent to which\n", + "the resulting fits differ. Such an approach may allow us to obtain\n", + "information that would not be available from fitting the model only\n", + "once using the original training sample.\n", + "\n", + "Two resampling methods are often used in Machine Learning analyses,\n", + "1. The **bootstrap method**\n", + "\n", + "2. and **Cross-Validation**\n", + "\n", + "In addition there are several other methods such as the Jackknife and the Blocking methods. We will discuss in particular\n", + "cross-validation and the bootstrap method. \n", + "\n", + "\n", + "Resampling approaches can be computationally expensive, because they\n", + "involve fitting the same statistical method multiple times using\n", + "different subsets of the training data. However, due to recent\n", + "advances in computing power, the computational requirements of\n", + "resampling methods generally are not prohibitive. In this chapter, we\n", + "discuss two of the most commonly used resampling methods,\n", + "cross-validation and the bootstrap. Both methods are important tools\n", + "in the practical application of many statistical learning\n", + "procedures. For example, cross-validation can be used to estimate the\n", + "test error associated with a given statistical learning method in\n", + "order to evaluate its performance, or to select the appropriate level\n", + "of flexibility. The process of evaluating a model’s performance is\n", + "known as model assessment, whereas the process of selecting the proper\n", + "level of flexibility for a model is known as model selection. The\n", + "bootstrap is widely used.\n", + "\n", + "\n", + "* Our simulations can be treated as *computer experiments*. This is particularly the case for Monte Carlo methods\n", + "\n", + "* The results can be analysed with the same statistical tools as we would use analysing experimental data.\n", + "\n", + "* As in all experiments, we are looking for expectation values and an estimate of how accurate they are, i.e., possible sources for errors.\n", + "\n", + "## Reminder on Statistics\n", + "\n", + "\n", + "* As in other experiments, many numerical experiments have two classes of errors:\n", + "\n", + " * Statistical errors\n", + "\n", + " * Systematical errors\n", + "\n", + "\n", + "* Statistical errors can be estimated using standard tools from statistics\n", + "\n", + "* Systematical errors are method specific and must be treated differently from case to case. \n", + "\n", + "The\n", + "advantage of doing linear regression is that we actually end up with\n", + "analytical expressions for several statistical quantities. \n", + "Standard least squares and Ridge regression allow us to\n", + "derive quantities like the variance and other expectation values in a\n", + "rather straightforward way.\n", + "\n", + "\n", + "It is assumed that $\\varepsilon_i\n", + "\\sim \\mathcal{N}(0, \\sigma^2)$ and the $\\varepsilon_{i}$ are\n", + "independent, i.e.:" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*} \n", + "\\mbox{Cov}(\\varepsilon_{i_1},\n", + "\\varepsilon_{i_2}) & = \\left\\{ \\begin{array}{lcc} \\sigma^2 & \\mbox{if}\n", + "& i_1 = i_2, \\\\ 0 & \\mbox{if} & i_1 \\not= i_2. \\end{array} \\right.\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The randomness of $\\varepsilon_i$ implies that\n", + "$\\mathbf{y}_i$ is also a random variable. In particular,\n", + "$\\mathbf{y}_i$ is normally distributed, because $\\varepsilon_i \\sim\n", + "\\mathcal{N}(0, \\sigma^2)$ and $\\mathbf{X}_{i,\\ast} \\, \\boldsymbol{\\beta}$ is a\n", + "non-random scalar. To specify the parameters of the distribution of\n", + "$\\mathbf{y}_i$ we need to calculate its first two moments. \n", + "\n", + "Recall that $\\boldsymbol{X}$ is a matrix of dimensionality $n\\times p$. The\n", + "notation above $\\mathbf{X}_{i,\\ast}$ means that we are looking at the\n", + "row number $i$ and perform a sum over all values $p$.\n", + "\n", + "\n", + "The assumption we have made here can be summarized as (and this is going to be useful when we discuss the bias-variance trade off)\n", + "that there exists a function $f(\\boldsymbol{x})$ and a normal distributed error $\\boldsymbol{\\varepsilon}\\sim \\mathcal{N}(0, \\sigma^2)$\n", + "which describe our data" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y} = f(\\boldsymbol{x})+\\boldsymbol{\\varepsilon}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We approximate this function with our model from the solution of the linear regression equations, that is our\n", + "function $f$ is approximated by $\\boldsymbol{\\tilde{y}}$ where we want to minimize $(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2$, our MSE, with" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\beta}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can calculate the expectation value of $\\boldsymbol{y}$ for a given element $i$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*} \n", + "\\mathbb{E}(y_i) & =\n", + "\\mathbb{E}(\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}) + \\mathbb{E}(\\varepsilon_i)\n", + "\\, \\, \\, = \\, \\, \\, \\mathbf{X}_{i, \\ast} \\, \\beta, \n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "while\n", + "its variance is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*} \\mbox{Var}(y_i) & = \\mathbb{E} \\{ [y_i\n", + "- \\mathbb{E}(y_i)]^2 \\} \\, \\, \\, = \\, \\, \\, \\mathbb{E} ( y_i^2 ) -\n", + "[\\mathbb{E}(y_i)]^2 \\\\ & = \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\,\n", + "\\beta + \\varepsilon_i )^2] - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \\\\ &\n", + "= \\mathbb{E} [ ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2 \\varepsilon_i\n", + "\\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} + \\varepsilon_i^2 ] - ( \\mathbf{X}_{i,\n", + "\\ast} \\, \\beta)^2 \\\\ & = ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 + 2\n", + "\\mathbb{E}(\\varepsilon_i) \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta} +\n", + "\\mathbb{E}(\\varepsilon_i^2 ) - ( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta})^2 \n", + "\\\\ & = \\mathbb{E}(\\varepsilon_i^2 ) \\, \\, \\, = \\, \\, \\,\n", + "\\mbox{Var}(\\varepsilon_i) \\, \\, \\, = \\, \\, \\, \\sigma^2. \n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Hence, $y_i \\sim \\mathcal{N}( \\mathbf{X}_{i, \\ast} \\, \\boldsymbol{\\beta}, \\sigma^2)$, that is $\\boldsymbol{y}$ follows a normal distribution with \n", + "mean value $\\boldsymbol{X}\\boldsymbol{\\beta}$ and variance $\\sigma^2$ (not be confused with the singular values of the SVD). \n", + "\n", + "\n", + "With the OLS expressions for the parameters $\\boldsymbol{\\beta}$ we can evaluate the expectation value" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}(\\boldsymbol{\\beta}) = \\mathbb{E}[ (\\mathbf{X}^{\\top} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1}\\mathbf{X}^{T} \\mathbb{E}[ \\mathbf{Y}]=(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\mathbf{X}^{T}\\mathbf{X}\\boldsymbol{\\beta}=\\boldsymbol{\\beta}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This means that the estimator of the regression parameters is unbiased.\n", + "\n", + "We can also calculate the variance\n", + "\n", + "The variance of $\\boldsymbol{\\beta}$ is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{eqnarray*}\n", + "\\mbox{Var}(\\boldsymbol{\\beta}) & = & \\mathbb{E} \\{ [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})] [\\boldsymbol{\\beta} - \\mathbb{E}(\\boldsymbol{\\beta})]^{T} \\}\n", + "\\\\\n", + "& = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} - \\boldsymbol{\\beta}]^{T} \\}\n", + "\\\\\n", + "% & = & \\mathbb{E} \\{ [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}] \\, [(\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y}]^{T} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% \\\\\n", + "% & = & \\mathbb{E} \\{ (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\mathbf{Y} \\, \\mathbf{Y}^{T} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% \\\\\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\mathbb{E} \\{ \\mathbf{Y} \\, \\mathbf{Y}^{T} \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "\\\\\n", + "& = & (\\mathbf{X}^{T} \\mathbf{X})^{-1} \\, \\mathbf{X}^{T} \\, \\{ \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} + \\sigma^2 \\} \\, \\mathbf{X} \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "% \\\\\n", + "% & = & (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^T \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T % \\mathbf{X})^{-1}\n", + "% \\\\\n", + "% & & + \\, \\, \\sigma^2 \\, (\\mathbf{X}^T \\mathbf{X})^{-1} \\, \\mathbf{X}^T \\, \\mathbf{X} \\, (\\mathbf{X}^T \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\boldsymbol{\\beta}^T\n", + "\\\\\n", + "& = & \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} + \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1} - \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T}\n", + "\\, \\, \\, = \\, \\, \\, \\sigma^2 \\, (\\mathbf{X}^{T} \\mathbf{X})^{-1},\n", + "\\end{eqnarray*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have used that $\\mathbb{E} (\\mathbf{Y} \\mathbf{Y}^{T}) =\n", + "\\mathbf{X} \\, \\boldsymbol{\\beta} \\, \\boldsymbol{\\beta}^{T} \\, \\mathbf{X}^{T} +\n", + "\\sigma^2 \\, \\mathbf{I}_{nn}$. From $\\mbox{Var}(\\boldsymbol{\\beta}) = \\sigma^2\n", + "\\, (\\mathbf{X}^{T} \\mathbf{X})^{-1}$, one obtains an estimate of the\n", + "variance of the estimate of the $j$-th regression coefficient:\n", + "$\\boldsymbol{\\sigma}^2 (\\boldsymbol{\\beta}_j ) = \\boldsymbol{\\sigma}^2 \\sqrt{\n", + "[(\\mathbf{X}^{T} \\mathbf{X})^{-1}]_{jj} }$. This may be used to\n", + "construct a confidence interval for the estimates.\n", + "\n", + "\n", + "In a similar way, we can obtain analytical expressions for say the\n", + "expectation values of the parameters $\\boldsymbol{\\beta}$ and their variance\n", + "when we employ Ridge regression, allowing us again to define a confidence interval. \n", + "\n", + "It is rather straightforward to show that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big]=(\\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I}_{pp})^{-1} (\\mathbf{X}^{\\top} \\mathbf{X})\\boldsymbol{\\beta}^{\\mathrm{OLS}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We see clearly that \n", + "$\\mathbb{E} \\big[ \\boldsymbol{\\beta}^{\\mathrm{Ridge}} \\big] \\not= \\boldsymbol{\\beta}^{\\mathrm{OLS}}$ for any $\\lambda > 0$. We say then that the ridge estimator is biased.\n", + "\n", + "We can also compute the variance as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{Ridge}}]=\\sigma^2[ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1} \\mathbf{X}^{T} \\mathbf{X} \\{ [ \\mathbf{X}^{\\top} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and it is easy to see that if the parameter $\\lambda$ goes to infinity then the variance of Ridge parameters $\\boldsymbol{\\beta}$ goes to zero. \n", + "\n", + "With this, we can compute the difference" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mbox{Var}[\\boldsymbol{\\beta}^{\\mathrm{OLS}}]-\\mbox{Var}(\\boldsymbol{\\beta}^{\\mathrm{Ridge}})=\\sigma^2 [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}[ 2\\lambda\\mathbf{I} + \\lambda^2 (\\mathbf{X}^{T} \\mathbf{X})^{-1} ] \\{ [ \\mathbf{X}^{T} \\mathbf{X} + \\lambda \\mathbf{I} ]^{-1}\\}^{T}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The difference is non-negative definite since each component of the\n", + "matrix product is non-negative definite. \n", + "This means the variance we obtain with the standard OLS will always for $\\lambda > 0$ be larger than the variance of $\\boldsymbol{\\beta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below. \n", + "\n", + "\n", + "\n", + "## Resampling methods\n", + "\n", + "With all these analytical equations for both the OLS and Ridge\n", + "regression, we will now outline how to assess a given model. This will\n", + "lead us to a discussion of the so-called bias-variance tradeoff (see\n", + "below) and so-called resampling methods.\n", + "\n", + "One of the quantities we have discussed as a way to measure errors is\n", + "the mean-squared error (MSE), mainly used for fitting of continuous\n", + "functions. Another choice is the absolute error.\n", + "\n", + "In the discussions below we will focus on the MSE and in particular since we will split the data into test and training data,\n", + "we discuss the\n", + "1. prediction error or simply the **test error** $\\mathrm{Err_{Test}}$, where we have a fixed training set and the test error is the MSE arising from the data reserved for testing. We discuss also the \n", + "\n", + "2. training error $\\mathrm{Err_{Train}}$, which is the average loss over the training data.\n", + "\n", + "As our model becomes more and more complex, more of the training data tends to used. The training may thence adapt to more complicated structures in the data. This may lead to a decrease in the bias (see below for code example) and a slight increase of the variance for the test error.\n", + "For a certain level of complexity the test error will reach minimum, before starting to increase again. The\n", + "training error reaches a saturation.\n", + "\n", + "\n", + "\n", + "Two famous\n", + "resampling methods are the **independent bootstrap** and **the jackknife**. \n", + "\n", + "The jackknife is a special case of the independent bootstrap. Still, the jackknife was made\n", + "popular prior to the independent bootstrap. And as the popularity of\n", + "the independent bootstrap soared, new variants, such as **the dependent bootstrap**.\n", + "\n", + "The Jackknife and independent bootstrap work for\n", + "independent, identically distributed random variables.\n", + "If these conditions are not\n", + "satisfied, the methods will fail. Yet, it should be said that if the data are\n", + "independent, identically distributed, and we only want to estimate the\n", + "variance of $\\overline{X}$ (which often is the case), then there is no\n", + "need for bootstrapping. \n", + "\n", + "\n", + "The Jackknife works by making many replicas of the estimator $\\widehat{\\theta}$. \n", + "The jackknife is a resampling method where we systematically leave out one observation from the vector of observed values $\\boldsymbol{x} = (x_1,x_2,\\cdots,X_n)$. \n", + "Let $\\boldsymbol{x}_i$ denote the vector" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_i = (x_1,x_2,\\cdots,x_{i-1},x_{i+1},\\cdots,x_n),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which equals the vector $\\boldsymbol{x}$ with the exception that observation\n", + "number $i$ is left out. Using this notation, define\n", + "$\\widehat{\\theta}_i$ to be the estimator\n", + "$\\widehat{\\theta}$ computed using $\\vec{X}_i$." + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Runtime: 0.136239 sec\n", + "Jackknife Statistics :\n", + "original bias std. error\n", + " 99.9154 99.9054 0.150467\n" + ] + } + ], + "source": [ + "from numpy import *\n", + "from numpy.random import randint, randn\n", + "from time import time\n", + "\n", + "def jackknife(data, stat):\n", + " n = len(data);t = zeros(n); inds = arange(n); t0 = time()\n", + " ## 'jackknifing' by leaving out an observation for each i \n", + " for i in range(n):\n", + " t[i] = stat(delete(data,i) )\n", + "\n", + " # analysis \n", + " print(\"Runtime: %g sec\" % (time()-t0)); print(\"Jackknife Statistics :\")\n", + " print(\"original bias std. error\")\n", + " print(\"%8g %14g %15g\" % (stat(data),(n-1)*mean(t)/n, (n*var(t))**.5))\n", + "\n", + " return t\n", + "\n", + "\n", + "# Returns mean of data samples \n", + "def stat(data):\n", + " return mean(data)\n", + "\n", + "\n", + "mu, sigma = 100, 15\n", + "datapoints = 10000\n", + "x = mu + sigma*random.randn(datapoints)\n", + "# jackknife returns the data sample \n", + "t = jackknife(x, stat)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### Bootstrap\n", + "\n", + "Bootstrapping is a nonparametric approach to statistical inference\n", + "that substitutes computation for more traditional distributional\n", + "assumptions and asymptotic results. Bootstrapping offers a number of\n", + "advantages: \n", + "1. The bootstrap is quite general, although there are some cases in which it fails. \n", + "\n", + "2. Because it does not require distributional assumptions (such as normally distributed errors), the bootstrap can provide more accurate inferences when the data are not well behaved or when the sample size is small. \n", + "\n", + "3. It is possible to apply the bootstrap to statistics with sampling distributions that are difficult to derive, even asymptotically. \n", + "\n", + "4. It is relatively simple to apply the bootstrap to complex data-collection plans (such as stratified and clustered samples).\n", + "\n", + "Since $\\widehat{\\theta} = \\widehat{\\theta}(\\boldsymbol{X})$ is a function of random variables,\n", + "$\\widehat{\\theta}$ itself must be a random variable. Thus it has\n", + "a pdf, call this function $p(\\boldsymbol{t})$. The aim of the bootstrap is to\n", + "estimate $p(\\boldsymbol{t})$ by the relative frequency of\n", + "$\\widehat{\\theta}$. You can think of this as using a histogram\n", + "in the place of $p(\\boldsymbol{t})$. If the relative frequency closely\n", + "resembles $p(\\vec{t})$, then using numerics, it is straight forward to\n", + "estimate all the interesting parameters of $p(\\boldsymbol{t})$ using point\n", + "estimators. \n", + "\n", + "\n", + "\n", + "In the case that $\\widehat{\\theta}$ has\n", + "more than one component, and the components are independent, we use the\n", + "same estimator on each component separately. If the probability\n", + "density function of $X_i$, $p(x)$, had been known, then it would have\n", + "been straight forward to do this by: \n", + "1. Drawing lots of numbers from $p(x)$, suppose we call one such set of numbers $(X_1^*, X_2^*, \\cdots, X_n^*)$. \n", + "\n", + "2. Then using these numbers, we could compute a replica of $\\widehat{\\theta}$ called $\\widehat{\\theta}^*$. \n", + "\n", + "By repeated use of (1) and (2), many\n", + "estimates of $\\widehat{\\theta}$ could have been obtained. The\n", + "idea is to use the relative frequency of $\\widehat{\\theta}^*$\n", + "(think of a histogram) as an estimate of $p(\\boldsymbol{t})$.\n", + "\n", + "\n", + "But\n", + "unless there is enough information available about the process that\n", + "generated $X_1,X_2,\\cdots,X_n$, $p(x)$ is in general\n", + "unknown. Therefore, [Efron in 1979](https://projecteuclid.org/euclid.aos/1176344552) asked the\n", + "question: What if we replace $p(x)$ by the relative frequency\n", + "of the observation $X_i$; if we draw observations in accordance with\n", + "the relative frequency of the observations, will we obtain the same\n", + "result in some asymptotic sense? The answer is yes.\n", + "\n", + "\n", + "Instead of generating the histogram for the relative\n", + "frequency of the observation $X_i$, just draw the values\n", + "$(X_1^*,X_2^*,\\cdots,X_n^*)$ with replacement from the vector\n", + "$\\boldsymbol{X}$. \n", + "\n", + "\n", + "The independent bootstrap works like this: \n", + "\n", + "1. Draw with replacement $n$ numbers for the observed variables $\\boldsymbol{x} = (x_1,x_2,\\cdots,x_n)$. \n", + "\n", + "2. Define a vector $\\boldsymbol{x}^*$ containing the values which were drawn from $\\boldsymbol{x}$. \n", + "\n", + "3. Using the vector $\\boldsymbol{x}^*$ compute $\\widehat{\\theta}^*$ by evaluating $\\widehat \\theta$ under the observations $\\boldsymbol{x}^*$. \n", + "\n", + "4. Repeat this process $k$ times. \n", + "\n", + "When you are done, you can draw a histogram of the relative frequency\n", + "of $\\widehat \\theta^*$. This is your estimate of the probability\n", + "distribution $p(t)$. Using this probability distribution you can\n", + "estimate any statistics thereof. In principle you never draw the\n", + "histogram of the relative frequency of $\\widehat{\\theta}^*$. Instead\n", + "you use the estimators corresponding to the statistic of interest. For\n", + "example, if you are interested in estimating the variance of $\\widehat\n", + "\\theta$, apply the etsimator $\\widehat \\sigma^2$ to the values\n", + "$\\widehat \\theta ^*$.\n", + "\n", + "\n", + "\n", + "The following code starts with a Gaussian distribution with mean value\n", + "$\\mu =100$ and variance $\\sigma=15$. We use this to generate the data\n", + "used in the bootstrap analysis. The bootstrap analysis returns a data\n", + "set after a given number of bootstrap operations (as many as we have\n", + "data points). This data set consists of estimated mean values for each\n", + "bootstrap operation. The histogram generated by the bootstrap method\n", + "shows that the distribution for these mean values is also a Gaussian,\n", + "centered around the mean value $\\mu=100$ but with standard deviation\n", + "$\\sigma/\\sqrt{n}$, where $n$ is the number of bootstrap samples (in\n", + "this case the same as the number of original data points). The value\n", + "of the standard deviation is what we expect from the central limit\n", + "theorem." + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Runtime: 1.77593 sec\n", + "Bootstrap Statistics :\n", + "original bias std. error\n", + " 100.207 14.9646 100.209 0.150205\n" + ] + }, + { + "ename": "AttributeError", + "evalue": "'Rectangle' object has no property 'normed'", + "output_type": "error", + "traceback": [ + "\u001b[0;31m---------------------------------------------------------------------------\u001b[0m", + "\u001b[0;31mAttributeError\u001b[0m Traceback (most recent call last)", + "\u001b[0;32m\u001b[0m in \u001b[0;36m\u001b[0;34m\u001b[0m\n\u001b[1;32m 31\u001b[0m \u001b[0mt\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mbootstrap\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mx\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mstat\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mdatapoints\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 32\u001b[0m \u001b[0;31m# the histogram of the bootstrapped data\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 33\u001b[0;31m \u001b[0mn\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mbinsboot\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mpatches\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mplt\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mhist\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mt\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;36m50\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mnormed\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;36m1\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mfacecolor\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m'red'\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0malpha\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;36m0.75\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 34\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 35\u001b[0m \u001b[0;31m# add a 'best fit' line\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", + "\u001b[0;32m~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/pyplot.py\u001b[0m in \u001b[0;36mhist\u001b[0;34m(x, bins, range, density, weights, cumulative, bottom, histtype, align, orientation, rwidth, log, color, label, stacked, data, **kwargs)\u001b[0m\n\u001b[1;32m 2683\u001b[0m \u001b[0morientation\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m'vertical'\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mrwidth\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;32mNone\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mlog\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;32mFalse\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mcolor\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;32mNone\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 2684\u001b[0m label=None, stacked=False, *, data=None, **kwargs):\n\u001b[0;32m-> 2685\u001b[0;31m return gca().hist(\n\u001b[0m\u001b[1;32m 2686\u001b[0m \u001b[0mx\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mbins\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mbins\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mrange\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mrange\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mdensity\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mdensity\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mweights\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mweights\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 2687\u001b[0m \u001b[0mcumulative\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mcumulative\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mbottom\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mbottom\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mhisttype\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mhisttype\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", + "\u001b[0;32m~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/__init__.py\u001b[0m in \u001b[0;36minner\u001b[0;34m(ax, data, *args, **kwargs)\u001b[0m\n\u001b[1;32m 1445\u001b[0m \u001b[0;32mdef\u001b[0m \u001b[0minner\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0max\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m*\u001b[0m\u001b[0margs\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mdata\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;32mNone\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 1446\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0mdata\u001b[0m \u001b[0;32mis\u001b[0m \u001b[0;32mNone\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 1447\u001b[0;31m \u001b[0;32mreturn\u001b[0m \u001b[0mfunc\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0max\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m*\u001b[0m\u001b[0mmap\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0msanitize_sequence\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0margs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 1448\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 1449\u001b[0m \u001b[0mbound\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mnew_sig\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mbind\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0max\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m*\u001b[0m\u001b[0margs\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", + "\u001b[0;32m~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/axes/_axes.py\u001b[0m in \u001b[0;36mhist\u001b[0;34m(self, x, bins, range, density, weights, cumulative, bottom, histtype, align, orientation, rwidth, log, color, label, stacked, **kwargs)\u001b[0m\n\u001b[1;32m 6813\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0mpatch\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 6814\u001b[0m \u001b[0mp\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mpatch\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;36m0\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 6815\u001b[0;31m \u001b[0mp\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mupdate\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 6816\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0mlbl\u001b[0m \u001b[0;32mis\u001b[0m \u001b[0;32mnot\u001b[0m \u001b[0;32mNone\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 6817\u001b[0m \u001b[0mp\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mset_label\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mlbl\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", + "\u001b[0;32m~/opt/anaconda3/lib/python3.8/site-packages/matplotlib/artist.py\u001b[0m in \u001b[0;36mupdate\u001b[0;34m(self, props)\u001b[0m\n\u001b[1;32m 994\u001b[0m \u001b[0mfunc\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mgetattr\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34mf\"set_{k}\"\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;32mNone\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 995\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0;32mnot\u001b[0m \u001b[0mcallable\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mfunc\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 996\u001b[0;31m raise AttributeError(f\"{type(self).__name__!r} object \"\n\u001b[0m\u001b[1;32m 997\u001b[0m f\"has no property {k!r}\")\n\u001b[1;32m 998\u001b[0m \u001b[0mret\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mappend\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mfunc\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mv\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", + "\u001b[0;31mAttributeError\u001b[0m: 'Rectangle' object has no property 'normed'" + ] + }, + { + "data": { + "image/png": "iVBORw0KGgoAAAANSUhEUgAAAXcAAAD4CAYAAAAXUaZHAAAAOXRFWHRTb2Z0d2FyZQBNYXRwbG90bGliIHZlcnNpb24zLjMuMywgaHR0cHM6Ly9tYXRwbG90bGliLm9yZy/Il7ecAAAACXBIWXMAAAsTAAALEwEAmpwYAAAR80lEQVR4nO3df7BcZ33f8fcnNoaGEIRtRSMkOfIUpQyT1j+4A6akGYIKg00aedLggWlrlSjRpDWtSTINSn9MJ5P8Yc+kJfYk40SDiOVMAjgEjxXiobgCmmYKhOvEEf4Fll07kipbF2I7oZ40mPn2j31U1uq9unvv3d179dz3a2Znz3nOc84+j/b6s4+fPedsqgpJUl++Y7UbIEkaP8NdkjpkuEtShwx3SeqQ4S5JHTp/tRsAcPHFF9f27dtXuxmSdE657777vlZVG+fbtibCffv27czOzq52MyTpnJLkyYW2OS0jSR0y3CWpQ4a7JHXIcJekDhnuktQhw12SOmS4S1KHDHdJ6pDhLkkdWhNXqErLtX3fH8xb/sRN75xyS6S1xZG7JHXIcJekDhnuktQhw12SOmS4S1KHRgr3JBuSfDzJI0keTvKmJBcmuTfJo+35Va1uktya5GiSI0munGwXJElnGnXkfgvwqap6LXAZ8DCwDzhcVTuAw20d4GpgR3vsBW4ba4slSYtaNNyTvBL4QeAAQFX9TVU9C+wCDrZqB4Fr2/Iu4I4a+AKwIcnmMbdbknQWo1zEdCkwB/xmksuA+4AbgU1VdbLVeQrY1Ja3AMeG9j/eyk4irTIvetJ6Mcq0zPnAlcBtVXUF8L/59hQMAFVVQC3lhZPsTTKbZHZubm4pu0qSFjFKuB8HjlfVF9v6xxmE/dOnp1va86m2/QSwbWj/ra3sRapqf1XNVNXMxo3z/ni3JGmZFg33qnoKOJbk77SincBDwCFgdyvbDdzdlg8B17ezZq4CnhuavpEkTcGoNw77V8BvJ7kAeBx4L4MPhjuT7AGeBK5rde8BrgGOAs+3utJULTS3Lq0XI4V7Vd0PzMyzaec8dQu4YWXNUu/8YlOaLG/5qzXF0JfGw9sPSFKHDHdJ6pDhLkkdMtwlqUN+oapzwqRPbfSLXPXGkbskdchwl6QOGe6S1CHDXZI6ZLhLUocMd0nqkOEuSR0y3CWpQ4a7JHXIcJekDhnuktQhw12SOmS4S1KHDHdJ6pDhLkkdMtwlqUOGuyR1yHCXpA6N9DN7SZ4A/gr4FvBCVc0kuRD4GLAdeAK4rqqeSRLgFuAa4Hngn1fVn4y/6dLkne3n/fwJPq1lSxm5/1BVXV5VM219H3C4qnYAh9s6wNXAjvbYC9w2rsZKkkazkmmZXcDBtnwQuHao/I4a+AKwIcnmFbyOJGmJRg33Aj6d5L4ke1vZpqo62ZafAja15S3AsaF9j7eyF0myN8lsktm5ubllNF2StJCR5tyBH6iqE0m+B7g3ySPDG6uqktRSXriq9gP7AWZmZpa0ryTp7EYauVfVifZ8CrgLeAPw9OnplvZ8qlU/AWwb2n1rK5MkTcmi4Z7k5UlecXoZeDvwAHAI2N2q7QbubsuHgOszcBXw3ND0jSRpCkaZltkE3DU4w5Hzgd+pqk8l+RJwZ5I9wJPAda3+PQxOgzzK4FTI94691ZKks1o03KvqceCyecq/Duycp7yAG8bSOknSsniFqiR1yHCXpA6NeiqkpDMsdGsCb0ugtcCRuyR1yHCXpA4Z7pLUIcNdkjrkF6qaqLPdD13S5Dhyl6QOGe6S1CHDXZI6ZLhLUocMd0nqkOEuSR0y3CWpQ4a7JHXIcJekDhnuktQhw12SOuS9ZTQW3kNGWlscuUtShwx3SeqQ4S5JHTLcJalDI4d7kvOS/GmST7b1S5N8McnRJB9LckErf2lbP9q2b59Q2yVJC1jKyP1G4OGh9ZuBD1bVa4BngD2tfA/wTCv/YKsnSZqikcI9yVbgncCH2nqAtwIfb1UOAte25V1tnbZ9Z6svSZqSUc9z/xXg54BXtPWLgGer6oW2fhzY0pa3AMcAquqFJM+1+l8bPmCSvcBegEsuuWSZzZfWnoXO+X/ipndOuSVazxYduSf5YeBUVd03zheuqv1VNVNVMxs3bhznoSVp3Rtl5P5m4EeSXAO8DPhu4BZgQ5Lz2+h9K3Ci1T8BbAOOJzkfeCXw9bG3XJK0oEVH7lX181W1taq2A+8GPlNV/wT4LPBjrdpu4O62fKit07Z/pqpqrK2WJJ3VSs5z/wDwM0mOMphTP9DKDwAXtfKfAfatrImSpKVa0o3DqupzwOfa8uPAG+ap89fAu8bQNknSMnmFqiR1yHCXpA4Z7pLUIcNdkjpkuEtSh/yZPS2JP6cnnRscuUtShwx3SeqQ4S5JHTLcJalDhrskdcizZaQp8Uc8NE2O3CWpQ4a7JHXIcJekDhnuktQhw12SOmS4S1KHDHdJ6pDhLkkdMtwlqUOGuyR1yNsPSKvM2xJoEhy5S1KHFg33JC9L8sdJ/izJg0l+oZVfmuSLSY4m+ViSC1r5S9v60bZ9+4T7IEk6wygj9/8DvLWqLgMuB96R5CrgZuCDVfUa4BlgT6u/B3imlX+w1ZMkTdGi4V4D32irL2mPAt4KfLyVHwSubcu72jpt+84kGVeDJUmLG2nOPcl5Se4HTgH3Ao8Bz1bVC63KcWBLW94CHANo258DLprnmHuTzCaZnZubW1EnJEkvNtLZMlX1LeDyJBuAu4DXrvSFq2o/sB9gZmamVno8jddCZ3BIOjcs6WyZqnoW+CzwJmBDktMfDluBE235BLANoG1/JfD1cTRWkjSaRUfuSTYC36yqZ5P8LeBtDL4k/SzwY8BHgd3A3W2XQ2398237Z6rKkbm0RJ7/rpUYZVpmM3AwyXkMRvp3VtUnkzwEfDTJLwF/Chxo9Q8Av5XkKPAXwLsn0G5J0lksGu5VdQS4Yp7yx4E3zFP+18C7xtI6SdKyeIWqJHXIcJekDhnuktQhw12SOmS4S1KHvJ/7OuZVqFK/HLlLUocMd0nqkOEuSR0y3CWpQ4a7JHXIcJekDhnuktQhw12SOmS4S1KHDHdJ6pDhLkkdMtwlqUOGuyR1yHCXpA4Z7pLUIcNdkjpkuEtShwx3SerQouGeZFuSzyZ5KMmDSW5s5RcmuTfJo+35Va08SW5NcjTJkSRXTroTkqQXG2Xk/gLws1X1OuAq4IYkrwP2AYeragdwuK0DXA3saI+9wG1jb7Uk6awW/YHsqjoJnGzLf5XkYWALsAt4S6t2EPgc8IFWfkdVFfCFJBuSbG7H0Srwh7D7stD7+cRN75xyS7SWLWnOPcl24Argi8CmocB+CtjUlrcAx4Z2O97KzjzW3iSzSWbn5uaW2m5J0lmMHO5Jvgv4PeD9VfWXw9vaKL2W8sJVtb+qZqpqZuPGjUvZVZK0iJHCPclLGAT7b1fVJ1rx00k2t+2bgVOt/ASwbWj3ra1MkjQli865JwlwAHi4qv7z0KZDwG7gpvZ891D5+5J8FHgj8Jzz7dPh3Lqk0xYNd+DNwD8Dvpzk/lb2bxmE+p1J9gBPAte1bfcA1wBHgeeB946zwZKkxY1ytswfAVlg88556hdwwwrbJUlaAa9QlaQOjTItI+kc4PnvGubIXZI6ZLhLUocMd0nqkOEuSR0y3CWpQ4a7JHXIcJekDhnuktQhw12SOmS4S1KHDHdJ6pDhLkkdMtwlqUOGuyR1yHCXpA55P3epc97nfX1y5C5JHXLkfg5aaCQmSac5cpekDhnuktQhw12SOmS4S1KHFg33JB9OcirJA0NlFya5N8mj7flVrTxJbk1yNMmRJFdOsvGSpPmNcrbM7cCvAncMle0DDlfVTUn2tfUPAFcDO9rjjcBt7VnSGuP5731bdOReVX8I/MUZxbuAg235IHDtUPkdNfAFYEOSzWNqqyRpRMudc99UVSfb8lPApra8BTg2VO94K/v/JNmbZDbJ7Nzc3DKbIUmaz4ovYqqqSlLL2G8/sB9gZmZmyfuvB16sJGm5ljtyf/r0dEt7PtXKTwDbhuptbWWSpClabrgfAna35d3A3UPl17ezZq4CnhuavpEkTcmi0zJJPgK8Bbg4yXHgPwI3AXcm2QM8CVzXqt8DXAMcBZ4H3juBNkuSFrFouFfVexbYtHOeugXcsNJGrTfOrUsaN69QlaQOGe6S1CHDXZI6ZLhLUof8JSZJL+I9Z/rgyF2SOmS4S1KHDHdJ6pDhLkkdMtwlqUOGuyR1yFMhp8h7yEiaFsN9Agxx9ehsf9eeA7/2OC0jSR0y3CWpQ4a7JHXIOXdJK+b9aNYew30F/OJU0lpluI/AEJd0rnHOXZI65Mhd0sQ4F796DHdJU2foT57TMpLUoXU3cvcSaknrQapq/AdN3gHcApwHfKiqbjpb/ZmZmZqdnR1rGzzDRerfeh+QJbmvqmbm2zb2kXuS84BfA94GHAe+lORQVT007tcCQ1xaz5y7X9gkpmXeABytqscBknwU2AVMJNwl6UxLDf1xDRLP9qEy7Q+iSYT7FuDY0Ppx4I1nVkqyF9jbVr+R5CsTaMu4XAx8bbUbMWb26dzRY79WpU+5eaKHvzg3L71PK2zT9y60YdW+UK2q/cD+1Xr9pUgyu9C81rnKPp07euyXfZq8SZwKeQLYNrS+tZVJkqZkEuH+JWBHkkuTXAC8Gzg0gdeRJC1g7NMyVfVCkvcB/4XBqZAfrqoHx/06U3ZOTB8tkX06d/TYL/s0YRM5z12StLq8/YAkdchwl6QOretwT3JjkgeSPJjk/a3ssiSfT/LlJL+f5LsX2Pen234PJPlIkpdNtfEvbsuHk5xK8sBQ2YVJ7k3yaHt+VStPkluTHE1yJMmVCxzz9e3f4Girn2n1p73+WPuU5DuT/EGSR9r7dtZbYkzKJN6roeMcGj7utEzo7++CJPuTfLW9Z/94Wv1prz+JPr2n/Td1JMmnklw8yT6s23BP8v3ATzK4ovYy4IeTvAb4ELCvqv4ucBfwb+bZdwvwr4GZqvp+Bl8cv3tabZ/H7cA7zijbBxyuqh3A4bYOcDWwoz32ArctcMzbGPz7nK575vEn7fZ5XnOlffrlqnotcAXw5iRXj7vRI7id8feLJD8KfGPcjR3R7Yy/T/8OOFVV3we8DvhvY27zYm5njH1Kcj6D+239UFX9PeAI8L6JtPy0qlqXD+BdwIGh9f8A/BzwHN/+onkb8NA8+56+CvdCBmccfRJ4+yr3ZzvwwND6V4DNbXkz8JW2/BvAe+arN1S2GXhkaP09wG+cy32a59i3AD95rr9Xrfy7gD9iEIIPTKrdU+7TMeDlq9GXSfQJeAkwx+CK0gC/DuydZPvX7cgdeAD4B0kuSvKdwDUMwvxBBvfCgcEHwLYzd6yqE8AvA38OnASeq6pPT6XVo9tUVSfb8lPAprY83+0htpyx75ZWfrY6q2Elffp/kmwA/hGD0ddasNJ+/SLwn4DnJ9bCpVt2n9r7A/CLSf4kye8m2cTqW3afquqbwL8Avgz8LwYfxAcm2dh1G+5V9TBwM/Bp4FPA/cC3gB8H/mWS+4BXAH9z5r5trm0XcCnwauDlSf7pdFq+dDUYOnR1zuty+9T+9/gjwK3Vbm63liy1X0kuB/52Vd01sUat0DLeq/MZXNn+P6rqSuDzDAZTa8Yy3qeXMAj3KxhkxhHg5yfTuoF1G+4AVXWgql5fVT8IPAN8taoeqaq3V9XrGYTAY/Ps+g+B/1lVc+0T+RPA359ey0fydJLNAO35VCsf5fYQJ1r52eqshpX06bT9wKNV9SuTauQyrKRfbwJmkjzBYGrm+5J8bqKtHc1K+vR1Bv8X8om2/rvAWb9MnpKV9OlygKp6rH0w3MmEM2Ndh3uS72nPlwA/CvzOUNl3AP+ewdzYmf4cuKqdgRFgJ/DwdFo9skPA7ra8G7h7qPz69g3/VQymlE4O79jW/zLJVa1/1w/tv5qW3SeAJL8EvBJ4/xTauhQrea9uq6pXV9V24AcYDFDeMp1mn9VK+lTA7wNvaUU7WRu3DF/J398J4HVJNrb1tzHpzFjNLyxW+wH8dwZ/NH8G7GxlNwJfbY+b+PaXq68G7hna9xeARxjM3f8W8NJV7MdHGMz9f5PBfN8e4CIGc8qPAv8VuLDVDYMfU3mMwfzfzNBx7h9anml9ewz41dP/DudqnxiMporBf1D3t8dP9PBeDZVtZxW+UJ3Q39/3An/IYPriMHBJB336qfb3d4TBh9dFk+yDtx+QpA6t62kZSeqV4S5JHTLcJalDhrskdchwl6QOGe6S1CHDXZI69H8BkyPDNx9dgJoAAAAASUVORK5CYII=\n", + "text/plain": [ + "
" + ] + }, + "metadata": { + "filenames": { + "image/png": "/Users/mhjensen/Teaching/MachineLearning/doc/LectureNotes/_build/jupyter_execute/chapter2_25_2.png" + }, + "needs_background": "light" + }, + "output_type": "display_data" + } + ], + "source": [ + "%matplotlib inline\n", + "\n", + "from numpy import *\n", + "from numpy.random import randint, randn\n", + "from time import time\n", + "import matplotlib.mlab as mlab\n", + "import matplotlib.pyplot as plt\n", + "\n", + "# Returns mean of bootstrap samples \n", + "def stat(data):\n", + " return mean(data)\n", + "\n", + "# Bootstrap algorithm\n", + "def bootstrap(data, statistic, R):\n", + " t = zeros(R); n = len(data); inds = arange(n); t0 = time()\n", + " # non-parametric bootstrap \n", + " for i in range(R):\n", + " t[i] = statistic(data[randint(0,n,n)])\n", + "\n", + " # analysis \n", + " print(\"Runtime: %g sec\" % (time()-t0)); print(\"Bootstrap Statistics :\")\n", + " print(\"original bias std. error\")\n", + " print(\"%8g %8g %14g %15g\" % (statistic(data), std(data),mean(t),std(t)))\n", + " return t\n", + "\n", + "\n", + "mu, sigma = 100, 15\n", + "datapoints = 10000\n", + "x = mu + sigma*random.randn(datapoints)\n", + "# bootstrap returns the data sample \n", + "t = bootstrap(x, stat, datapoints)\n", + "# the histogram of the bootstrapped data \n", + "n, binsboot, patches = plt.hist(t, 50, normed=1, facecolor='red', alpha=0.75)\n", + "\n", + "# add a 'best fit' line \n", + "y = mlab.normpdf( binsboot, mean(t), std(t))\n", + "lt = plt.plot(binsboot, y, 'r--', linewidth=1)\n", + "plt.xlabel('Smarts')\n", + "plt.ylabel('Probability')\n", + "plt.axis([99.5, 100.6, 0, 3.0])\n", + "plt.grid(True)\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Various steps in cross-validation\n", + "\n", + "When the repetitive splitting of the data set is done randomly,\n", + "samples may accidently end up in a fast majority of the splits in\n", + "either training or test set. Such samples may have an unbalanced\n", + "influence on either model building or prediction evaluation. To avoid\n", + "this $k$-fold cross-validation structures the data splitting. The\n", + "samples are divided into $k$ more or less equally sized exhaustive and\n", + "mutually exclusive subsets. In turn (at each split) one of these\n", + "subsets plays the role of the test set while the union of the\n", + "remaining subsets constitutes the training set. Such a splitting\n", + "warrants a balanced representation of each sample in both training and\n", + "test set over the splits. Still the division into the $k$ subsets\n", + "involves a degree of randomness. This may be fully excluded when\n", + "choosing $k=n$. This particular case is referred to as leave-one-out\n", + "cross-validation (LOOCV). \n", + "\n", + "\n", + "* Define a range of interest for the penalty parameter.\n", + "\n", + "* Divide the data set into training and test set comprising samples $\\{1, \\ldots, n\\} \\setminus i$ and $\\{ i \\}$, respectively.\n", + "\n", + "* Fit the linear regression model by means of ridge estimation for each $\\lambda$ in the grid using the training set, and the corresponding estimate of the error variance $\\boldsymbol{\\sigma}_{-i}^2(\\lambda)$, as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "\\boldsymbol{\\beta}_{-i}(\\lambda) & = ( \\boldsymbol{X}_{-i, \\ast}^{T}\n", + "\\boldsymbol{X}_{-i, \\ast} + \\lambda \\boldsymbol{I}_{pp})^{-1}\n", + "\\boldsymbol{X}_{-i, \\ast}^{T} \\boldsymbol{y}_{-i}\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "* Evaluate the prediction performance of these models on the test set by $\\log\\{L[y_i, \\boldsymbol{X}_{i, \\ast}; \\boldsymbol{\\beta}_{-i}(\\lambda), \\boldsymbol{\\sigma}_{-i}^2(\\lambda)]\\}$. Or, by the prediction error $|y_i - \\boldsymbol{X}_{i, \\ast} \\boldsymbol{\\beta}_{-i}(\\lambda)|$, the relative error, the error squared or the R2 score function.\n", + "\n", + "* Repeat the first three steps such that each sample plays the role of the test set once.\n", + "\n", + "* Average the prediction performances of the test sets at each grid point of the penalty bias/parameter. It is an estimate of the prediction performance of the model corresponding to this value of the penalty parameter on novel data. It is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "\\frac{1}{n} \\sum_{i = 1}^n \\log\\{L[y_i, \\mathbf{X}_{i, \\ast}; \\boldsymbol{\\beta}_{-i}(\\lambda), \\boldsymbol{\\sigma}_{-i}^2(\\lambda)]\\}.\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "For the various values of $k$\n", + "\n", + "1. shuffle the dataset randomly.\n", + "\n", + "2. Split the dataset into $k$ groups.\n", + "\n", + "3. For each unique group:\n", + "\n", + "a. Decide which group to use as set for test data\n", + "\n", + "b. Take the remaining groups as a training data set\n", + "\n", + "c. Fit a model on the training set and evaluate it on the test set\n", + "\n", + "d. Retain the evaluation score and discard the model\n", + "\n", + "\n", + "5. Summarize the model using the sample of model evaluation scores\n", + "\n", + "The code here uses Ridge regression with cross-validation (CV) resampling and $k$-fold CV in order to fit a specific polynomial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import KFold\n", + "from sklearn.linear_model import Ridge\n", + "from sklearn.model_selection import cross_val_score\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "\n", + "# A seed just to ensure that the random numbers are the same for every run.\n", + "# Useful for eventual debugging.\n", + "np.random.seed(3155)\n", + "\n", + "# Generate the data.\n", + "nsamples = 100\n", + "x = np.random.randn(nsamples)\n", + "y = 3*x**2 + np.random.randn(nsamples)\n", + "\n", + "## Cross-validation on Ridge regression using KFold only\n", + "\n", + "# Decide degree on polynomial to fit\n", + "poly = PolynomialFeatures(degree = 6)\n", + "\n", + "# Decide which values of lambda to use\n", + "nlambdas = 500\n", + "lambdas = np.logspace(-3, 5, nlambdas)\n", + "\n", + "# Initialize a KFold instance\n", + "k = 5\n", + "kfold = KFold(n_splits = k)\n", + "\n", + "# Perform the cross-validation to estimate MSE\n", + "scores_KFold = np.zeros((nlambdas, k))\n", + "\n", + "i = 0\n", + "for lmb in lambdas:\n", + " ridge = Ridge(alpha = lmb)\n", + " j = 0\n", + " for train_inds, test_inds in kfold.split(x):\n", + " xtrain = x[train_inds]\n", + " ytrain = y[train_inds]\n", + "\n", + " xtest = x[test_inds]\n", + " ytest = y[test_inds]\n", + "\n", + " Xtrain = poly.fit_transform(xtrain[:, np.newaxis])\n", + " ridge.fit(Xtrain, ytrain[:, np.newaxis])\n", + "\n", + " Xtest = poly.fit_transform(xtest[:, np.newaxis])\n", + " ypred = ridge.predict(Xtest)\n", + "\n", + " scores_KFold[i,j] = np.sum((ypred - ytest[:, np.newaxis])**2)/np.size(ypred)\n", + "\n", + " j += 1\n", + " i += 1\n", + "\n", + "\n", + "estimated_mse_KFold = np.mean(scores_KFold, axis = 1)\n", + "\n", + "## Cross-validation using cross_val_score from sklearn along with KFold\n", + "\n", + "# kfold is an instance initialized above as:\n", + "# kfold = KFold(n_splits = k)\n", + "\n", + "estimated_mse_sklearn = np.zeros(nlambdas)\n", + "i = 0\n", + "for lmb in lambdas:\n", + " ridge = Ridge(alpha = lmb)\n", + "\n", + " X = poly.fit_transform(x[:, np.newaxis])\n", + " estimated_mse_folds = cross_val_score(ridge, X, y[:, np.newaxis], scoring='neg_mean_squared_error', cv=kfold)\n", + "\n", + " # cross_val_score return an array containing the estimated negative mse for every fold.\n", + " # we have to the the mean of every array in order to get an estimate of the mse of the model\n", + " estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)\n", + "\n", + " i += 1\n", + "\n", + "## Plot and compare the slightly different ways to perform cross-validation\n", + "\n", + "plt.figure()\n", + "\n", + "plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')\n", + "plt.plot(np.log10(lambdas), estimated_mse_KFold, 'r--', label = 'KFold')\n", + "\n", + "plt.xlabel('log10(lambda)')\n", + "plt.ylabel('mse')\n", + "\n", + "plt.legend()\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## The bias-variance tradeoff\n", + "\n", + "\n", + "We will discuss the bias-variance tradeoff in the context of\n", + "continuous predictions such as regression. However, many of the\n", + "intuitions and ideas discussed here also carry over to classification\n", + "tasks. Consider a dataset $\\mathcal{L}$ consisting of the data\n", + "$\\mathbf{X}_\\mathcal{L}=\\{(y_j, \\boldsymbol{x}_j), j=0\\ldots n-1\\}$. \n", + "\n", + "Let us assume that the true data is generated from a noisy model" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y}=f(\\boldsymbol{x}) + \\boldsymbol{\\epsilon}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\epsilon$ is normally distributed with mean zero and standard deviation $\\sigma^2$.\n", + "\n", + "In our derivation of the ordinary least squares method we defined then\n", + "an approximation to the function $f$ in terms of the parameters\n", + "$\\boldsymbol{\\beta}$ and the design matrix $\\boldsymbol{X}$ which embody our model,\n", + "that is $\\boldsymbol{\\tilde{y}}=\\boldsymbol{X}\\boldsymbol{\\beta}$. \n", + "\n", + "Thereafter we found the parameters $\\boldsymbol{\\beta}$ by optimizing the means squared error via the so-called cost function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{X},\\boldsymbol{\\beta}) =\\frac{1}{n}\\sum_{i=0}^{n-1}(y_i-\\tilde{y}_i)^2=\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right].\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can rewrite this as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right]=\\frac{1}{n}\\sum_i(f_i-\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right])^2+\\frac{1}{n}\\sum_i(\\tilde{y}_i-\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right])^2+\\sigma^2.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The three terms represent the square of the bias of the learning\n", + "method, which can be thought of as the error caused by the simplifying\n", + "assumptions built into the method. The second term represents the\n", + "variance of the chosen model and finally the last terms is variance of\n", + "the error $\\boldsymbol{\\epsilon}$.\n", + "\n", + "To derive this equation, we need to recall that the variance of $\\boldsymbol{y}$ and $\\boldsymbol{\\epsilon}$ are both equal to $\\sigma^2$. The mean value of $\\boldsymbol{\\epsilon}$ is by definition equal to zero. Furthermore, the function $f$ is not a stochastics variable, idem for $\\boldsymbol{\\tilde{y}}$.\n", + "We use a more compact notation in terms of the expectation value" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right]=\\mathbb{E}\\left[(\\boldsymbol{f}+\\boldsymbol{\\epsilon}-\\boldsymbol{\\tilde{y}})^2\\right],\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and adding and subtracting $\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right]$ we get" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right]=\\mathbb{E}\\left[(\\boldsymbol{f}+\\boldsymbol{\\epsilon}-\\boldsymbol{\\tilde{y}}+\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right]-\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right])^2\\right],\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which, using the abovementioned expectation values can be rewritten as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}\\left[(\\boldsymbol{y}-\\boldsymbol{\\tilde{y}})^2\\right]=\\mathbb{E}\\left[(\\boldsymbol{y}-\\mathbb{E}\\left[\\boldsymbol{\\tilde{y}}\\right])^2\\right]+\\mathrm{Var}\\left[\\boldsymbol{\\tilde{y}}\\right]+\\sigma^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "that is the rewriting in terms of the so-called bias, the variance of the model $\\boldsymbol{\\tilde{y}}$ and the variance of $\\boldsymbol{\\epsilon}$." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.pipeline import make_pipeline\n", + "from sklearn.utils import resample\n", + "\n", + "np.random.seed(2018)\n", + "\n", + "n = 500\n", + "n_boostraps = 100\n", + "degree = 18 # A quite high value, just to show.\n", + "noise = 0.1\n", + "\n", + "# Make data set.\n", + "x = np.linspace(-1, 3, n).reshape(-1, 1)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) + np.random.normal(0, 0.1, x.shape)\n", + "\n", + "# Hold out some test data that is never used in training.\n", + "x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2)\n", + "\n", + "# Combine x transformation and model into one operation.\n", + "# Not neccesary, but convenient.\n", + "model = make_pipeline(PolynomialFeatures(degree=degree), LinearRegression(fit_intercept=False))\n", + "\n", + "# The following (m x n_bootstraps) matrix holds the column vectors y_pred\n", + "# for each bootstrap iteration.\n", + "y_pred = np.empty((y_test.shape[0], n_boostraps))\n", + "for i in range(n_boostraps):\n", + " x_, y_ = resample(x_train, y_train)\n", + "\n", + " # Evaluate the new model on the same test data each time.\n", + " y_pred[:, i] = model.fit(x_, y_).predict(x_test).ravel()\n", + "\n", + "# Note: Expectations and variances taken w.r.t. different training\n", + "# data sets, hence the axis=1. Subsequent means are taken across the test data\n", + "# set in order to obtain a total value, but before this we have error/bias/variance\n", + "# calculated per data point in the test set.\n", + "# Note 2: The use of keepdims=True is important in the calculation of bias as this \n", + "# maintains the column vector form. Dropping this yields very unexpected results.\n", + "error = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) )\n", + "bias = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 )\n", + "variance = np.mean( np.var(y_pred, axis=1, keepdims=True) )\n", + "print('Error:', error)\n", + "print('Bias^2:', bias)\n", + "print('Var:', variance)\n", + "print('{} >= {} + {} = {}'.format(error, bias, variance, bias+variance))\n", + "\n", + "plt.plot(x[::5, :], y[::5, :], label='f(x)')\n", + "plt.scatter(x_test, y_test, label='Data points')\n", + "plt.scatter(x_test, np.mean(y_pred, axis=1), label='Pred')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.pipeline import make_pipeline\n", + "from sklearn.utils import resample\n", + "\n", + "np.random.seed(2018)\n", + "\n", + "n = 40\n", + "n_boostraps = 100\n", + "maxdegree = 14\n", + "\n", + "\n", + "# Make data set.\n", + "x = np.linspace(-3, 3, n).reshape(-1, 1)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)\n", + "error = np.zeros(maxdegree)\n", + "bias = np.zeros(maxdegree)\n", + "variance = np.zeros(maxdegree)\n", + "polydegree = np.zeros(maxdegree)\n", + "x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2)\n", + "\n", + "for degree in range(maxdegree):\n", + " model = make_pipeline(PolynomialFeatures(degree=degree), LinearRegression(fit_intercept=False))\n", + " y_pred = np.empty((y_test.shape[0], n_boostraps))\n", + " for i in range(n_boostraps):\n", + " x_, y_ = resample(x_train, y_train)\n", + " y_pred[:, i] = model.fit(x_, y_).predict(x_test).ravel()\n", + "\n", + " polydegree[degree] = degree\n", + " error[degree] = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) )\n", + " bias[degree] = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 )\n", + " variance[degree] = np.mean( np.var(y_pred, axis=1, keepdims=True) )\n", + " print('Polynomial degree:', degree)\n", + " print('Error:', error[degree])\n", + " print('Bias^2:', bias[degree])\n", + " print('Var:', variance[degree])\n", + " print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree]))\n", + "\n", + "plt.plot(polydegree, error, label='Error')\n", + "plt.plot(polydegree, bias, label='bias')\n", + "plt.plot(polydegree, variance, label='Variance')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The bias-variance tradeoff summarizes the fundamental tension in\n", + "machine learning, particularly supervised learning, between the\n", + "complexity of a model and the amount of training data needed to train\n", + "it. Since data is often limited, in practice it is often useful to\n", + "use a less-complex model with higher bias, that is a model whose asymptotic\n", + "performance is worse than another model because it is easier to\n", + "train and less sensitive to sampling noise arising from having a\n", + "finite-sized training dataset (smaller variance). \n", + "\n", + "\n", + "\n", + "The above equations tell us that in\n", + "order to minimize the expected test error, we need to select a\n", + "statistical learning method that simultaneously achieves low variance\n", + "and low bias. Note that variance is inherently a nonnegative quantity,\n", + "and squared bias is also nonnegative. Hence, we see that the expected\n", + "test MSE can never lie below $Var(\\epsilon)$, the irreducible error.\n", + "\n", + "\n", + "What do we mean by the variance and bias of a statistical learning\n", + "method? The variance refers to the amount by which our model would change if we\n", + "estimated it using a different training data set. Since the training\n", + "data are used to fit the statistical learning method, different\n", + "training data sets will result in a different estimate. But ideally the\n", + "estimate for our model should not vary too much between training\n", + "sets. However, if a method has high variance then small changes in\n", + "the training data can result in large changes in the model. In general, more\n", + "flexible statistical methods have higher variance.\n", + "\n", + "\n", + "You may also find this recent [article](https://www.pnas.org/content/116/32/15849) of interest." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\"\"\"\n", + "============================\n", + "Underfitting vs. Overfitting\n", + "============================\n", + "\n", + "This example demonstrates the problems of underfitting and overfitting and\n", + "how we can use linear regression with polynomial features to approximate\n", + "nonlinear functions. The plot shows the function that we want to approximate,\n", + "which is a part of the cosine function. In addition, the samples from the\n", + "real function and the approximations of different models are displayed. The\n", + "models have polynomial features of different degrees. We can see that a\n", + "linear function (polynomial with degree 1) is not sufficient to fit the\n", + "training samples. This is called **underfitting**. A polynomial of degree 4\n", + "approximates the true function almost perfectly. However, for higher degrees\n", + "the model will **overfit** the training data, i.e. it learns the noise of the\n", + "training data.\n", + "We evaluate quantitatively **overfitting** / **underfitting** by using\n", + "cross-validation. We calculate the mean squared error (MSE) on the validation\n", + "set, the higher, the less likely the model generalizes correctly from the\n", + "training data.\n", + "\"\"\"\n", + "\n", + "print(__doc__)\n", + "\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.pipeline import Pipeline\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "from sklearn.linear_model import LinearRegression\n", + "from sklearn.model_selection import cross_val_score\n", + "\n", + "\n", + "def true_fun(X):\n", + " return np.cos(1.5 * np.pi * X)\n", + "\n", + "np.random.seed(0)\n", + "\n", + "n_samples = 30\n", + "degrees = [1, 4, 15]\n", + "\n", + "X = np.sort(np.random.rand(n_samples))\n", + "y = true_fun(X) + np.random.randn(n_samples) * 0.1\n", + "\n", + "plt.figure(figsize=(14, 5))\n", + "for i in range(len(degrees)):\n", + " ax = plt.subplot(1, len(degrees), i + 1)\n", + " plt.setp(ax, xticks=(), yticks=())\n", + "\n", + " polynomial_features = PolynomialFeatures(degree=degrees[i],\n", + " include_bias=False)\n", + " linear_regression = LinearRegression()\n", + " pipeline = Pipeline([(\"polynomial_features\", polynomial_features),\n", + " (\"linear_regression\", linear_regression)])\n", + " pipeline.fit(X[:, np.newaxis], y)\n", + "\n", + " # Evaluate the models using crossvalidation\n", + " scores = cross_val_score(pipeline, X[:, np.newaxis], y,\n", + " scoring=\"neg_mean_squared_error\", cv=10)\n", + "\n", + " X_test = np.linspace(0, 1, 100)\n", + " plt.plot(X_test, pipeline.predict(X_test[:, np.newaxis]), label=\"Model\")\n", + " plt.plot(X_test, true_fun(X_test), label=\"True function\")\n", + " plt.scatter(X, y, edgecolor='b', s=20, label=\"Samples\")\n", + " plt.xlabel(\"x\")\n", + " plt.ylabel(\"y\")\n", + " plt.xlim((0, 1))\n", + " plt.ylim((-2, 2))\n", + " plt.legend(loc=\"best\")\n", + " plt.title(\"Degree {}\\nMSE = {:.2e}(+/- {:.2e})\".format(\n", + " degrees[i], -scores.mean(), scores.std()))\n", + "plt.show()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.utils import resample\n", + "from sklearn.metrics import mean_squared_error\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"EoS.csv\"),'r')\n", + "\n", + "# Read the EoS data as csv file and organize the data into two arrays with density and energies\n", + "EoS = pd.read_csv(infile, names=('Density', 'Energy'))\n", + "EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')\n", + "EoS = EoS.dropna()\n", + "Energies = EoS['Energy']\n", + "Density = EoS['Density']\n", + "# The design matrix now as function of various polytrops\n", + "\n", + "Maxpolydegree = 30\n", + "X = np.zeros((len(Density),Maxpolydegree))\n", + "X[:,0] = 1.0\n", + "testerror = np.zeros(Maxpolydegree)\n", + "trainingerror = np.zeros(Maxpolydegree)\n", + "polynomial = np.zeros(Maxpolydegree)\n", + "\n", + "trials = 100\n", + "for polydegree in range(1, Maxpolydegree):\n", + " polynomial[polydegree] = polydegree\n", + " for degree in range(polydegree):\n", + " X[:,degree] = Density**(degree/3.0)\n", + "\n", + "# loop over trials in order to estimate the expectation value of the MSE\n", + " testerror[polydegree] = 0.0\n", + " trainingerror[polydegree] = 0.0\n", + " for samples in range(trials):\n", + " x_train, x_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2)\n", + " model = LinearRegression(fit_intercept=True).fit(x_train, y_train)\n", + " ypred = model.predict(x_train)\n", + " ytilde = model.predict(x_test)\n", + " testerror[polydegree] += mean_squared_error(y_test, ytilde)\n", + " trainingerror[polydegree] += mean_squared_error(y_train, ypred) \n", + "\n", + " testerror[polydegree] /= trials\n", + " trainingerror[polydegree] /= trials\n", + " print(\"Degree of polynomial: %3d\"% polynomial[polydegree])\n", + " print(\"Mean squared error on training data: %.8f\" % trainingerror[polydegree])\n", + " print(\"Mean squared error on test data: %.8f\" % testerror[polydegree])\n", + "\n", + "plt.plot(polynomial, np.log10(trainingerror), label='Training Error')\n", + "plt.plot(polynomial, np.log10(testerror), label='Test Error')\n", + "plt.xlabel('Polynomial degree')\n", + "plt.ylabel('log10[MSE]')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.metrics import mean_squared_error\n", + "from sklearn.model_selection import KFold\n", + "from sklearn.model_selection import cross_val_score\n", + "\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"EoS.csv\"),'r')\n", + "\n", + "# Read the EoS data as csv file and organize the data into two arrays with density and energies\n", + "EoS = pd.read_csv(infile, names=('Density', 'Energy'))\n", + "EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce')\n", + "EoS = EoS.dropna()\n", + "Energies = EoS['Energy']\n", + "Density = EoS['Density']\n", + "# The design matrix now as function of various polytrops\n", + "\n", + "Maxpolydegree = 30\n", + "X = np.zeros((len(Density),Maxpolydegree))\n", + "X[:,0] = 1.0\n", + "estimated_mse_sklearn = np.zeros(Maxpolydegree)\n", + "polynomial = np.zeros(Maxpolydegree)\n", + "k =5\n", + "kfold = KFold(n_splits = k)\n", + "\n", + "for polydegree in range(1, Maxpolydegree):\n", + " polynomial[polydegree] = polydegree\n", + " for degree in range(polydegree):\n", + " X[:,degree] = Density**(degree/3.0)\n", + " OLS = LinearRegression()\n", + "# loop over trials in order to estimate the expectation value of the MSE\n", + " estimated_mse_folds = cross_val_score(OLS, X, Energies, scoring='neg_mean_squared_error', cv=kfold)\n", + "#[:, np.newaxis]\n", + " estimated_mse_sklearn[polydegree] = np.mean(-estimated_mse_folds)\n", + "\n", + "plt.plot(polynomial, np.log10(estimated_mse_sklearn), label='Test Error')\n", + "plt.xlabel('Polynomial degree')\n", + "plt.ylabel('log10[MSE]')\n", + "plt.legend()\n", + "plt.show()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.model_selection import KFold\n", + "from sklearn.linear_model import Ridge\n", + "from sklearn.model_selection import cross_val_score\n", + "from sklearn.preprocessing import PolynomialFeatures\n", + "\n", + "# A seed just to ensure that the random numbers are the same for every run.\n", + "np.random.seed(3155)\n", + "# Generate the data.\n", + "n = 100\n", + "x = np.linspace(-3, 3, n).reshape(-1, 1)\n", + "y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)\n", + "# Decide degree on polynomial to fit\n", + "poly = PolynomialFeatures(degree = 10)\n", + "\n", + "# Decide which values of lambda to use\n", + "nlambdas = 500\n", + "lambdas = np.logspace(-3, 5, nlambdas)\n", + "# Initialize a KFold instance\n", + "k = 5\n", + "kfold = KFold(n_splits = k)\n", + "estimated_mse_sklearn = np.zeros(nlambdas)\n", + "i = 0\n", + "for lmb in lambdas:\n", + " ridge = Ridge(alpha = lmb)\n", + " estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)\n", + " estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)\n", + " i += 1\n", + "plt.figure()\n", + "plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')\n", + "plt.xlabel('log10(lambda)')\n", + "plt.ylabel('MSE')\n", + "plt.legend()\n", + "plt.show()" + ] + } + ], + "metadata": { + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} \ No newline at end of file diff --git a/doc/LectureNotes/_build/jupyter_execute/chapter2.py b/doc/LectureNotes/_build/jupyter_execute/chapter2.py new file mode 100644 index 000000000..f0b72ab51 --- /dev/null +++ b/doc/LectureNotes/_build/jupyter_execute/chapter2.py @@ -0,0 +1,1042 @@ +# Resampling Methods + +[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSept3.mp4?vrtx=view-as-webpage) + + +## Introduction + +Resampling methods are an indispensable tool in modern +statistics. They involve repeatedly drawing samples from a training +set and refitting a model of interest on each sample in order to +obtain additional information about the fitted model. For example, in +order to estimate the variability of a linear regression fit, we can +repeatedly draw different samples from the training data, fit a linear +regression to each new sample, and then examine the extent to which +the resulting fits differ. Such an approach may allow us to obtain +information that would not be available from fitting the model only +once using the original training sample. + +Two resampling methods are often used in Machine Learning analyses, +1. The **bootstrap method** + +2. and **Cross-Validation** + +In addition there are several other methods such as the Jackknife and the Blocking methods. We will discuss in particular +cross-validation and the bootstrap method. + + +Resampling approaches can be computationally expensive, because they +involve fitting the same statistical method multiple times using +different subsets of the training data. However, due to recent +advances in computing power, the computational requirements of +resampling methods generally are not prohibitive. In this chapter, we +discuss two of the most commonly used resampling methods, +cross-validation and the bootstrap. Both methods are important tools +in the practical application of many statistical learning +procedures. For example, cross-validation can be used to estimate the +test error associated with a given statistical learning method in +order to evaluate its performance, or to select the appropriate level +of flexibility. The process of evaluating a model’s performance is +known as model assessment, whereas the process of selecting the proper +level of flexibility for a model is known as model selection. The +bootstrap is widely used. + + +* Our simulations can be treated as *computer experiments*. This is particularly the case for Monte Carlo methods + +* The results can be analysed with the same statistical tools as we would use analysing experimental data. + +* As in all experiments, we are looking for expectation values and an estimate of how accurate they are, i.e., possible sources for errors. + +## Reminder on Statistics + + +* As in other experiments, many numerical experiments have two classes of errors: + + * Statistical errors + + * Systematical errors + + +* Statistical errors can be estimated using standard tools from statistics + +* Systematical errors are method specific and must be treated differently from case to case. + +The +advantage of doing linear regression is that we actually end up with +analytical expressions for several statistical quantities. +Standard least squares and Ridge regression allow us to +derive quantities like the variance and other expectation values in a +rather straightforward way. + + +It is assumed that $\varepsilon_i +\sim \mathcal{N}(0, \sigma^2)$ and the $\varepsilon_{i}$ are +independent, i.e.: + +$$ +\begin{align*} +\mbox{Cov}(\varepsilon_{i_1}, +\varepsilon_{i_2}) & = \left\{ \begin{array}{lcc} \sigma^2 & \mbox{if} +& i_1 = i_2, \\ 0 & \mbox{if} & i_1 \not= i_2. \end{array} \right. +\end{align*} +$$ + +The randomness of $\varepsilon_i$ implies that +$\mathbf{y}_i$ is also a random variable. In particular, +$\mathbf{y}_i$ is normally distributed, because $\varepsilon_i \sim +\mathcal{N}(0, \sigma^2)$ and $\mathbf{X}_{i,\ast} \, \boldsymbol{\beta}$ is a +non-random scalar. To specify the parameters of the distribution of +$\mathbf{y}_i$ we need to calculate its first two moments. + +Recall that $\boldsymbol{X}$ is a matrix of dimensionality $n\times p$. The +notation above $\mathbf{X}_{i,\ast}$ means that we are looking at the +row number $i$ and perform a sum over all values $p$. + + +The assumption we have made here can be summarized as (and this is going to be useful when we discuss the bias-variance trade off) +that there exists a function $f(\boldsymbol{x})$ and a normal distributed error $\boldsymbol{\varepsilon}\sim \mathcal{N}(0, \sigma^2)$ +which describe our data + +$$ +\boldsymbol{y} = f(\boldsymbol{x})+\boldsymbol{\varepsilon} +$$ + +We approximate this function with our model from the solution of the linear regression equations, that is our +function $f$ is approximated by $\boldsymbol{\tilde{y}}$ where we want to minimize $(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2$, our MSE, with + +$$ +\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\beta}. +$$ + +We can calculate the expectation value of $\boldsymbol{y}$ for a given element $i$ + +$$ +\begin{align*} +\mathbb{E}(y_i) & = +\mathbb{E}(\mathbf{X}_{i, \ast} \, \boldsymbol{\beta}) + \mathbb{E}(\varepsilon_i) +\, \, \, = \, \, \, \mathbf{X}_{i, \ast} \, \beta, +\end{align*} +$$ + +while +its variance is + +$$ +\begin{align*} \mbox{Var}(y_i) & = \mathbb{E} \{ [y_i +- \mathbb{E}(y_i)]^2 \} \, \, \, = \, \, \, \mathbb{E} ( y_i^2 ) - +[\mathbb{E}(y_i)]^2 \\ & = \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, +\beta + \varepsilon_i )^2] - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 \\ & += \mathbb{E} [ ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 \varepsilon_i +\mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + \varepsilon_i^2 ] - ( \mathbf{X}_{i, +\ast} \, \beta)^2 \\ & = ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 + 2 +\mathbb{E}(\varepsilon_i) \mathbf{X}_{i, \ast} \, \boldsymbol{\beta} + +\mathbb{E}(\varepsilon_i^2 ) - ( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta})^2 +\\ & = \mathbb{E}(\varepsilon_i^2 ) \, \, \, = \, \, \, +\mbox{Var}(\varepsilon_i) \, \, \, = \, \, \, \sigma^2. +\end{align*} +$$ + +Hence, $y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \boldsymbol{\beta}, \sigma^2)$, that is $\boldsymbol{y}$ follows a normal distribution with +mean value $\boldsymbol{X}\boldsymbol{\beta}$ and variance $\sigma^2$ (not be confused with the singular values of the SVD). + + +With the OLS expressions for the parameters $\boldsymbol{\beta}$ we can evaluate the expectation value + +$$ +\mathbb{E}(\boldsymbol{\beta}) = \mathbb{E}[ (\mathbf{X}^{\top} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1}\mathbf{X}^{T} \mathbb{E}[ \mathbf{Y}]=(\mathbf{X}^{T} \mathbf{X})^{-1} \mathbf{X}^{T}\mathbf{X}\boldsymbol{\beta}=\boldsymbol{\beta}. +$$ + +This means that the estimator of the regression parameters is unbiased. + +We can also calculate the variance + +The variance of $\boldsymbol{\beta}$ is + +$$ +\begin{eqnarray*} +\mbox{Var}(\boldsymbol{\beta}) & = & \mathbb{E} \{ [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})] [\boldsymbol{\beta} - \mathbb{E}(\boldsymbol{\beta})]^{T} \} +\\ +& = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} - \boldsymbol{\beta}]^{T} \} +\\ +% & = & \mathbb{E} \{ [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}] \, [(\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y}]^{T} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% \\ +% & = & \mathbb{E} \{ (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \mathbf{Y} \, \mathbf{Y}^{T} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} \} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% \\ +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \mathbb{E} \{ \mathbf{Y} \, \mathbf{Y}^{T} \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +\\ +& = & (\mathbf{X}^{T} \mathbf{X})^{-1} \, \mathbf{X}^{T} \, \{ \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + \sigma^2 \} \, \mathbf{X} \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +% \\ +% & = & (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^T \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T % \mathbf{X})^{-1} +% \\ +% & & + \, \, \sigma^2 \, (\mathbf{X}^T \mathbf{X})^{-1} \, \mathbf{X}^T \, \mathbf{X} \, (\mathbf{X}^T \mathbf{X})^{-1} - \boldsymbol{\beta} \boldsymbol{\beta}^T +\\ +& = & \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} + \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1} - \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} +\, \, \, = \, \, \, \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}, +\end{eqnarray*} +$$ + +where we have used that $\mathbb{E} (\mathbf{Y} \mathbf{Y}^{T}) = +\mathbf{X} \, \boldsymbol{\beta} \, \boldsymbol{\beta}^{T} \, \mathbf{X}^{T} + +\sigma^2 \, \mathbf{I}_{nn}$. From $\mbox{Var}(\boldsymbol{\beta}) = \sigma^2 +\, (\mathbf{X}^{T} \mathbf{X})^{-1}$, one obtains an estimate of the +variance of the estimate of the $j$-th regression coefficient: +$\boldsymbol{\sigma}^2 (\boldsymbol{\beta}_j ) = \boldsymbol{\sigma}^2 \sqrt{ +[(\mathbf{X}^{T} \mathbf{X})^{-1}]_{jj} }$. This may be used to +construct a confidence interval for the estimates. + + +In a similar way, we can obtain analytical expressions for say the +expectation values of the parameters $\boldsymbol{\beta}$ and their variance +when we employ Ridge regression, allowing us again to define a confidence interval. + +It is rather straightforward to show that + +$$ +\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big]=(\mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I}_{pp})^{-1} (\mathbf{X}^{\top} \mathbf{X})\boldsymbol{\beta}^{\mathrm{OLS}}. +$$ + +We see clearly that +$\mathbb{E} \big[ \boldsymbol{\beta}^{\mathrm{Ridge}} \big] \not= \boldsymbol{\beta}^{\mathrm{OLS}}$ for any $\lambda > 0$. We say then that the ridge estimator is biased. + +We can also compute the variance as + +$$ +\mbox{Var}[\boldsymbol{\beta}^{\mathrm{Ridge}}]=\sigma^2[ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1} \mathbf{X}^{T} \mathbf{X} \{ [ \mathbf{X}^{\top} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}, +$$ + +and it is easy to see that if the parameter $\lambda$ goes to infinity then the variance of Ridge parameters $\boldsymbol{\beta}$ goes to zero. + +With this, we can compute the difference + +$$ +\mbox{Var}[\boldsymbol{\beta}^{\mathrm{OLS}}]-\mbox{Var}(\boldsymbol{\beta}^{\mathrm{Ridge}})=\sigma^2 [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}[ 2\lambda\mathbf{I} + \lambda^2 (\mathbf{X}^{T} \mathbf{X})^{-1} ] \{ [ \mathbf{X}^{T} \mathbf{X} + \lambda \mathbf{I} ]^{-1}\}^{T}. +$$ + +The difference is non-negative definite since each component of the +matrix product is non-negative definite. +This means the variance we obtain with the standard OLS will always for $\lambda > 0$ be larger than the variance of $\boldsymbol{\beta}$ obtained with the Ridge estimator. This has interesting consequences when we discuss the so-called bias-variance trade-off below. + + + +## Resampling methods + +With all these analytical equations for both the OLS and Ridge +regression, we will now outline how to assess a given model. This will +lead us to a discussion of the so-called bias-variance tradeoff (see +below) and so-called resampling methods. + +One of the quantities we have discussed as a way to measure errors is +the mean-squared error (MSE), mainly used for fitting of continuous +functions. Another choice is the absolute error. + +In the discussions below we will focus on the MSE and in particular since we will split the data into test and training data, +we discuss the +1. prediction error or simply the **test error** $\mathrm{Err_{Test}}$, where we have a fixed training set and the test error is the MSE arising from the data reserved for testing. We discuss also the + +2. training error $\mathrm{Err_{Train}}$, which is the average loss over the training data. + +As our model becomes more and more complex, more of the training data tends to used. The training may thence adapt to more complicated structures in the data. This may lead to a decrease in the bias (see below for code example) and a slight increase of the variance for the test error. +For a certain level of complexity the test error will reach minimum, before starting to increase again. The +training error reaches a saturation. + + + +Two famous +resampling methods are the **independent bootstrap** and **the jackknife**. + +The jackknife is a special case of the independent bootstrap. Still, the jackknife was made +popular prior to the independent bootstrap. And as the popularity of +the independent bootstrap soared, new variants, such as **the dependent bootstrap**. + +The Jackknife and independent bootstrap work for +independent, identically distributed random variables. +If these conditions are not +satisfied, the methods will fail. Yet, it should be said that if the data are +independent, identically distributed, and we only want to estimate the +variance of $\overline{X}$ (which often is the case), then there is no +need for bootstrapping. + + +The Jackknife works by making many replicas of the estimator $\widehat{\theta}$. +The jackknife is a resampling method where we systematically leave out one observation from the vector of observed values $\boldsymbol{x} = (x_1,x_2,\cdots,X_n)$. +Let $\boldsymbol{x}_i$ denote the vector + +$$ +\boldsymbol{x}_i = (x_1,x_2,\cdots,x_{i-1},x_{i+1},\cdots,x_n), +$$ + +which equals the vector $\boldsymbol{x}$ with the exception that observation +number $i$ is left out. Using this notation, define +$\widehat{\theta}_i$ to be the estimator +$\widehat{\theta}$ computed using $\vec{X}_i$. + +from numpy import * +from numpy.random import randint, randn +from time import time + +def jackknife(data, stat): + n = len(data);t = zeros(n); inds = arange(n); t0 = time() + ## 'jackknifing' by leaving out an observation for each i + for i in range(n): + t[i] = stat(delete(data,i) ) + + # analysis + print("Runtime: %g sec" % (time()-t0)); print("Jackknife Statistics :") + print("original bias std. error") + print("%8g %14g %15g" % (stat(data),(n-1)*mean(t)/n, (n*var(t))**.5)) + + return t + + +# Returns mean of data samples +def stat(data): + return mean(data) + + +mu, sigma = 100, 15 +datapoints = 10000 +x = mu + sigma*random.randn(datapoints) +# jackknife returns the data sample +t = jackknife(x, stat) + +### Bootstrap + +Bootstrapping is a nonparametric approach to statistical inference +that substitutes computation for more traditional distributional +assumptions and asymptotic results. Bootstrapping offers a number of +advantages: +1. The bootstrap is quite general, although there are some cases in which it fails. + +2. Because it does not require distributional assumptions (such as normally distributed errors), the bootstrap can provide more accurate inferences when the data are not well behaved or when the sample size is small. + +3. It is possible to apply the bootstrap to statistics with sampling distributions that are difficult to derive, even asymptotically. + +4. It is relatively simple to apply the bootstrap to complex data-collection plans (such as stratified and clustered samples). + +Since $\widehat{\theta} = \widehat{\theta}(\boldsymbol{X})$ is a function of random variables, +$\widehat{\theta}$ itself must be a random variable. Thus it has +a pdf, call this function $p(\boldsymbol{t})$. The aim of the bootstrap is to +estimate $p(\boldsymbol{t})$ by the relative frequency of +$\widehat{\theta}$. You can think of this as using a histogram +in the place of $p(\boldsymbol{t})$. If the relative frequency closely +resembles $p(\vec{t})$, then using numerics, it is straight forward to +estimate all the interesting parameters of $p(\boldsymbol{t})$ using point +estimators. + + + +In the case that $\widehat{\theta}$ has +more than one component, and the components are independent, we use the +same estimator on each component separately. If the probability +density function of $X_i$, $p(x)$, had been known, then it would have +been straight forward to do this by: +1. Drawing lots of numbers from $p(x)$, suppose we call one such set of numbers $(X_1^*, X_2^*, \cdots, X_n^*)$. + +2. Then using these numbers, we could compute a replica of $\widehat{\theta}$ called $\widehat{\theta}^*$. + +By repeated use of (1) and (2), many +estimates of $\widehat{\theta}$ could have been obtained. The +idea is to use the relative frequency of $\widehat{\theta}^*$ +(think of a histogram) as an estimate of $p(\boldsymbol{t})$. + + +But +unless there is enough information available about the process that +generated $X_1,X_2,\cdots,X_n$, $p(x)$ is in general +unknown. Therefore, [Efron in 1979](https://projecteuclid.org/euclid.aos/1176344552) asked the +question: What if we replace $p(x)$ by the relative frequency +of the observation $X_i$; if we draw observations in accordance with +the relative frequency of the observations, will we obtain the same +result in some asymptotic sense? The answer is yes. + + +Instead of generating the histogram for the relative +frequency of the observation $X_i$, just draw the values +$(X_1^*,X_2^*,\cdots,X_n^*)$ with replacement from the vector +$\boldsymbol{X}$. + + +The independent bootstrap works like this: + +1. Draw with replacement $n$ numbers for the observed variables $\boldsymbol{x} = (x_1,x_2,\cdots,x_n)$. + +2. Define a vector $\boldsymbol{x}^*$ containing the values which were drawn from $\boldsymbol{x}$. + +3. Using the vector $\boldsymbol{x}^*$ compute $\widehat{\theta}^*$ by evaluating $\widehat \theta$ under the observations $\boldsymbol{x}^*$. + +4. Repeat this process $k$ times. + +When you are done, you can draw a histogram of the relative frequency +of $\widehat \theta^*$. This is your estimate of the probability +distribution $p(t)$. Using this probability distribution you can +estimate any statistics thereof. In principle you never draw the +histogram of the relative frequency of $\widehat{\theta}^*$. Instead +you use the estimators corresponding to the statistic of interest. For +example, if you are interested in estimating the variance of $\widehat +\theta$, apply the etsimator $\widehat \sigma^2$ to the values +$\widehat \theta ^*$. + + + +The following code starts with a Gaussian distribution with mean value +$\mu =100$ and variance $\sigma=15$. We use this to generate the data +used in the bootstrap analysis. The bootstrap analysis returns a data +set after a given number of bootstrap operations (as many as we have +data points). This data set consists of estimated mean values for each +bootstrap operation. The histogram generated by the bootstrap method +shows that the distribution for these mean values is also a Gaussian, +centered around the mean value $\mu=100$ but with standard deviation +$\sigma/\sqrt{n}$, where $n$ is the number of bootstrap samples (in +this case the same as the number of original data points). The value +of the standard deviation is what we expect from the central limit +theorem. + +%matplotlib inline + +from numpy import * +from numpy.random import randint, randn +from time import time +import matplotlib.mlab as mlab +import matplotlib.pyplot as plt + +# Returns mean of bootstrap samples +def stat(data): + return mean(data) + +# Bootstrap algorithm +def bootstrap(data, statistic, R): + t = zeros(R); n = len(data); inds = arange(n); t0 = time() + # non-parametric bootstrap + for i in range(R): + t[i] = statistic(data[randint(0,n,n)]) + + # analysis + print("Runtime: %g sec" % (time()-t0)); print("Bootstrap Statistics :") + print("original bias std. error") + print("%8g %8g %14g %15g" % (statistic(data), std(data),mean(t),std(t))) + return t + + +mu, sigma = 100, 15 +datapoints = 10000 +x = mu + sigma*random.randn(datapoints) +# bootstrap returns the data sample +t = bootstrap(x, stat, datapoints) +# the histogram of the bootstrapped data +n, binsboot, patches = plt.hist(t, 50, normed=1, facecolor='red', alpha=0.75) + +# add a 'best fit' line +y = mlab.normpdf( binsboot, mean(t), std(t)) +lt = plt.plot(binsboot, y, 'r--', linewidth=1) +plt.xlabel('Smarts') +plt.ylabel('Probability') +plt.axis([99.5, 100.6, 0, 3.0]) +plt.grid(True) + +plt.show() + +## Various steps in cross-validation + +When the repetitive splitting of the data set is done randomly, +samples may accidently end up in a fast majority of the splits in +either training or test set. Such samples may have an unbalanced +influence on either model building or prediction evaluation. To avoid +this $k$-fold cross-validation structures the data splitting. The +samples are divided into $k$ more or less equally sized exhaustive and +mutually exclusive subsets. In turn (at each split) one of these +subsets plays the role of the test set while the union of the +remaining subsets constitutes the training set. Such a splitting +warrants a balanced representation of each sample in both training and +test set over the splits. Still the division into the $k$ subsets +involves a degree of randomness. This may be fully excluded when +choosing $k=n$. This particular case is referred to as leave-one-out +cross-validation (LOOCV). + + +* Define a range of interest for the penalty parameter. + +* Divide the data set into training and test set comprising samples $\{1, \ldots, n\} \setminus i$ and $\{ i \}$, respectively. + +* Fit the linear regression model by means of ridge estimation for each $\lambda$ in the grid using the training set, and the corresponding estimate of the error variance $\boldsymbol{\sigma}_{-i}^2(\lambda)$, as + +$$ +\begin{align*} +\boldsymbol{\beta}_{-i}(\lambda) & = ( \boldsymbol{X}_{-i, \ast}^{T} +\boldsymbol{X}_{-i, \ast} + \lambda \boldsymbol{I}_{pp})^{-1} +\boldsymbol{X}_{-i, \ast}^{T} \boldsymbol{y}_{-i} +\end{align*} +$$ + +* Evaluate the prediction performance of these models on the test set by $\log\{L[y_i, \boldsymbol{X}_{i, \ast}; \boldsymbol{\beta}_{-i}(\lambda), \boldsymbol{\sigma}_{-i}^2(\lambda)]\}$. Or, by the prediction error $|y_i - \boldsymbol{X}_{i, \ast} \boldsymbol{\beta}_{-i}(\lambda)|$, the relative error, the error squared or the R2 score function. + +* Repeat the first three steps such that each sample plays the role of the test set once. + +* Average the prediction performances of the test sets at each grid point of the penalty bias/parameter. It is an estimate of the prediction performance of the model corresponding to this value of the penalty parameter on novel data. It is defined as + +$$ +\begin{align*} +\frac{1}{n} \sum_{i = 1}^n \log\{L[y_i, \mathbf{X}_{i, \ast}; \boldsymbol{\beta}_{-i}(\lambda), \boldsymbol{\sigma}_{-i}^2(\lambda)]\}. +\end{align*} +$$ + +For the various values of $k$ + +1. shuffle the dataset randomly. + +2. Split the dataset into $k$ groups. + +3. For each unique group: + +a. Decide which group to use as set for test data + +b. Take the remaining groups as a training data set + +c. Fit a model on the training set and evaluate it on the test set + +d. Retain the evaluation score and discard the model + + +5. Summarize the model using the sample of model evaluation scores + +The code here uses Ridge regression with cross-validation (CV) resampling and $k$-fold CV in order to fit a specific polynomial. + +import numpy as np +import matplotlib.pyplot as plt +from sklearn.model_selection import KFold +from sklearn.linear_model import Ridge +from sklearn.model_selection import cross_val_score +from sklearn.preprocessing import PolynomialFeatures + +# A seed just to ensure that the random numbers are the same for every run. +# Useful for eventual debugging. +np.random.seed(3155) + +# Generate the data. +nsamples = 100 +x = np.random.randn(nsamples) +y = 3*x**2 + np.random.randn(nsamples) + +## Cross-validation on Ridge regression using KFold only + +# Decide degree on polynomial to fit +poly = PolynomialFeatures(degree = 6) + +# Decide which values of lambda to use +nlambdas = 500 +lambdas = np.logspace(-3, 5, nlambdas) + +# Initialize a KFold instance +k = 5 +kfold = KFold(n_splits = k) + +# Perform the cross-validation to estimate MSE +scores_KFold = np.zeros((nlambdas, k)) + +i = 0 +for lmb in lambdas: + ridge = Ridge(alpha = lmb) + j = 0 + for train_inds, test_inds in kfold.split(x): + xtrain = x[train_inds] + ytrain = y[train_inds] + + xtest = x[test_inds] + ytest = y[test_inds] + + Xtrain = poly.fit_transform(xtrain[:, np.newaxis]) + ridge.fit(Xtrain, ytrain[:, np.newaxis]) + + Xtest = poly.fit_transform(xtest[:, np.newaxis]) + ypred = ridge.predict(Xtest) + + scores_KFold[i,j] = np.sum((ypred - ytest[:, np.newaxis])**2)/np.size(ypred) + + j += 1 + i += 1 + + +estimated_mse_KFold = np.mean(scores_KFold, axis = 1) + +## Cross-validation using cross_val_score from sklearn along with KFold + +# kfold is an instance initialized above as: +# kfold = KFold(n_splits = k) + +estimated_mse_sklearn = np.zeros(nlambdas) +i = 0 +for lmb in lambdas: + ridge = Ridge(alpha = lmb) + + X = poly.fit_transform(x[:, np.newaxis]) + estimated_mse_folds = cross_val_score(ridge, X, y[:, np.newaxis], scoring='neg_mean_squared_error', cv=kfold) + + # cross_val_score return an array containing the estimated negative mse for every fold. + # we have to the the mean of every array in order to get an estimate of the mse of the model + estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds) + + i += 1 + +## Plot and compare the slightly different ways to perform cross-validation + +plt.figure() + +plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score') +plt.plot(np.log10(lambdas), estimated_mse_KFold, 'r--', label = 'KFold') + +plt.xlabel('log10(lambda)') +plt.ylabel('mse') + +plt.legend() + +plt.show() + +## The bias-variance tradeoff + + +We will discuss the bias-variance tradeoff in the context of +continuous predictions such as regression. However, many of the +intuitions and ideas discussed here also carry over to classification +tasks. Consider a dataset $\mathcal{L}$ consisting of the data +$\mathbf{X}_\mathcal{L}=\{(y_j, \boldsymbol{x}_j), j=0\ldots n-1\}$. + +Let us assume that the true data is generated from a noisy model + +$$ +\boldsymbol{y}=f(\boldsymbol{x}) + \boldsymbol{\epsilon} +$$ + +where $\epsilon$ is normally distributed with mean zero and standard deviation $\sigma^2$. + +In our derivation of the ordinary least squares method we defined then +an approximation to the function $f$ in terms of the parameters +$\boldsymbol{\beta}$ and the design matrix $\boldsymbol{X}$ which embody our model, +that is $\boldsymbol{\tilde{y}}=\boldsymbol{X}\boldsymbol{\beta}$. + +Thereafter we found the parameters $\boldsymbol{\beta}$ by optimizing the means squared error via the so-called cost function + +$$ +C(\boldsymbol{X},\boldsymbol{\beta}) =\frac{1}{n}\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2=\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]. +$$ + +We can rewrite this as + +$$ +\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]=\frac{1}{n}\sum_i(f_i-\mathbb{E}\left[\boldsymbol{\tilde{y}}\right])^2+\frac{1}{n}\sum_i(\tilde{y}_i-\mathbb{E}\left[\boldsymbol{\tilde{y}}\right])^2+\sigma^2. +$$ + +The three terms represent the square of the bias of the learning +method, which can be thought of as the error caused by the simplifying +assumptions built into the method. The second term represents the +variance of the chosen model and finally the last terms is variance of +the error $\boldsymbol{\epsilon}$. + +To derive this equation, we need to recall that the variance of $\boldsymbol{y}$ and $\boldsymbol{\epsilon}$ are both equal to $\sigma^2$. The mean value of $\boldsymbol{\epsilon}$ is by definition equal to zero. Furthermore, the function $f$ is not a stochastics variable, idem for $\boldsymbol{\tilde{y}}$. +We use a more compact notation in terms of the expectation value + +$$ +\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]=\mathbb{E}\left[(\boldsymbol{f}+\boldsymbol{\epsilon}-\boldsymbol{\tilde{y}})^2\right], +$$ + +and adding and subtracting $\mathbb{E}\left[\boldsymbol{\tilde{y}}\right]$ we get + +$$ +\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]=\mathbb{E}\left[(\boldsymbol{f}+\boldsymbol{\epsilon}-\boldsymbol{\tilde{y}}+\mathbb{E}\left[\boldsymbol{\tilde{y}}\right]-\mathbb{E}\left[\boldsymbol{\tilde{y}}\right])^2\right], +$$ + +which, using the abovementioned expectation values can be rewritten as + +$$ +\mathbb{E}\left[(\boldsymbol{y}-\boldsymbol{\tilde{y}})^2\right]=\mathbb{E}\left[(\boldsymbol{y}-\mathbb{E}\left[\boldsymbol{\tilde{y}}\right])^2\right]+\mathrm{Var}\left[\boldsymbol{\tilde{y}}\right]+\sigma^2, +$$ + +that is the rewriting in terms of the so-called bias, the variance of the model $\boldsymbol{\tilde{y}}$ and the variance of $\boldsymbol{\epsilon}$. + +import matplotlib.pyplot as plt +import numpy as np +from sklearn.linear_model import LinearRegression, Ridge, Lasso +from sklearn.preprocessing import PolynomialFeatures +from sklearn.model_selection import train_test_split +from sklearn.pipeline import make_pipeline +from sklearn.utils import resample + +np.random.seed(2018) + +n = 500 +n_boostraps = 100 +degree = 18 # A quite high value, just to show. +noise = 0.1 + +# Make data set. +x = np.linspace(-1, 3, n).reshape(-1, 1) +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2) + np.random.normal(0, 0.1, x.shape) + +# Hold out some test data that is never used in training. +x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2) + +# Combine x transformation and model into one operation. +# Not neccesary, but convenient. +model = make_pipeline(PolynomialFeatures(degree=degree), LinearRegression(fit_intercept=False)) + +# The following (m x n_bootstraps) matrix holds the column vectors y_pred +# for each bootstrap iteration. +y_pred = np.empty((y_test.shape[0], n_boostraps)) +for i in range(n_boostraps): + x_, y_ = resample(x_train, y_train) + + # Evaluate the new model on the same test data each time. + y_pred[:, i] = model.fit(x_, y_).predict(x_test).ravel() + +# Note: Expectations and variances taken w.r.t. different training +# data sets, hence the axis=1. Subsequent means are taken across the test data +# set in order to obtain a total value, but before this we have error/bias/variance +# calculated per data point in the test set. +# Note 2: The use of keepdims=True is important in the calculation of bias as this +# maintains the column vector form. Dropping this yields very unexpected results. +error = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) ) +bias = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 ) +variance = np.mean( np.var(y_pred, axis=1, keepdims=True) ) +print('Error:', error) +print('Bias^2:', bias) +print('Var:', variance) +print('{} >= {} + {} = {}'.format(error, bias, variance, bias+variance)) + +plt.plot(x[::5, :], y[::5, :], label='f(x)') +plt.scatter(x_test, y_test, label='Data points') +plt.scatter(x_test, np.mean(y_pred, axis=1), label='Pred') +plt.legend() +plt.show() + +import matplotlib.pyplot as plt +import numpy as np +from sklearn.linear_model import LinearRegression, Ridge, Lasso +from sklearn.preprocessing import PolynomialFeatures +from sklearn.model_selection import train_test_split +from sklearn.pipeline import make_pipeline +from sklearn.utils import resample + +np.random.seed(2018) + +n = 40 +n_boostraps = 100 +maxdegree = 14 + + +# Make data set. +x = np.linspace(-3, 3, n).reshape(-1, 1) +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape) +error = np.zeros(maxdegree) +bias = np.zeros(maxdegree) +variance = np.zeros(maxdegree) +polydegree = np.zeros(maxdegree) +x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2) + +for degree in range(maxdegree): + model = make_pipeline(PolynomialFeatures(degree=degree), LinearRegression(fit_intercept=False)) + y_pred = np.empty((y_test.shape[0], n_boostraps)) + for i in range(n_boostraps): + x_, y_ = resample(x_train, y_train) + y_pred[:, i] = model.fit(x_, y_).predict(x_test).ravel() + + polydegree[degree] = degree + error[degree] = np.mean( np.mean((y_test - y_pred)**2, axis=1, keepdims=True) ) + bias[degree] = np.mean( (y_test - np.mean(y_pred, axis=1, keepdims=True))**2 ) + variance[degree] = np.mean( np.var(y_pred, axis=1, keepdims=True) ) + print('Polynomial degree:', degree) + print('Error:', error[degree]) + print('Bias^2:', bias[degree]) + print('Var:', variance[degree]) + print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree])) + +plt.plot(polydegree, error, label='Error') +plt.plot(polydegree, bias, label='bias') +plt.plot(polydegree, variance, label='Variance') +plt.legend() +plt.show() + +The bias-variance tradeoff summarizes the fundamental tension in +machine learning, particularly supervised learning, between the +complexity of a model and the amount of training data needed to train +it. Since data is often limited, in practice it is often useful to +use a less-complex model with higher bias, that is a model whose asymptotic +performance is worse than another model because it is easier to +train and less sensitive to sampling noise arising from having a +finite-sized training dataset (smaller variance). + + + +The above equations tell us that in +order to minimize the expected test error, we need to select a +statistical learning method that simultaneously achieves low variance +and low bias. Note that variance is inherently a nonnegative quantity, +and squared bias is also nonnegative. Hence, we see that the expected +test MSE can never lie below $Var(\epsilon)$, the irreducible error. + + +What do we mean by the variance and bias of a statistical learning +method? The variance refers to the amount by which our model would change if we +estimated it using a different training data set. Since the training +data are used to fit the statistical learning method, different +training data sets will result in a different estimate. But ideally the +estimate for our model should not vary too much between training +sets. However, if a method has high variance then small changes in +the training data can result in large changes in the model. In general, more +flexible statistical methods have higher variance. + + +You may also find this recent [article](https://www.pnas.org/content/116/32/15849) of interest. + +""" +============================ +Underfitting vs. Overfitting +============================ + +This example demonstrates the problems of underfitting and overfitting and +how we can use linear regression with polynomial features to approximate +nonlinear functions. The plot shows the function that we want to approximate, +which is a part of the cosine function. In addition, the samples from the +real function and the approximations of different models are displayed. The +models have polynomial features of different degrees. We can see that a +linear function (polynomial with degree 1) is not sufficient to fit the +training samples. This is called **underfitting**. A polynomial of degree 4 +approximates the true function almost perfectly. However, for higher degrees +the model will **overfit** the training data, i.e. it learns the noise of the +training data. +We evaluate quantitatively **overfitting** / **underfitting** by using +cross-validation. We calculate the mean squared error (MSE) on the validation +set, the higher, the less likely the model generalizes correctly from the +training data. +""" + +print(__doc__) + +import numpy as np +import matplotlib.pyplot as plt +from sklearn.pipeline import Pipeline +from sklearn.preprocessing import PolynomialFeatures +from sklearn.linear_model import LinearRegression +from sklearn.model_selection import cross_val_score + + +def true_fun(X): + return np.cos(1.5 * np.pi * X) + +np.random.seed(0) + +n_samples = 30 +degrees = [1, 4, 15] + +X = np.sort(np.random.rand(n_samples)) +y = true_fun(X) + np.random.randn(n_samples) * 0.1 + +plt.figure(figsize=(14, 5)) +for i in range(len(degrees)): + ax = plt.subplot(1, len(degrees), i + 1) + plt.setp(ax, xticks=(), yticks=()) + + polynomial_features = PolynomialFeatures(degree=degrees[i], + include_bias=False) + linear_regression = LinearRegression() + pipeline = Pipeline([("polynomial_features", polynomial_features), + ("linear_regression", linear_regression)]) + pipeline.fit(X[:, np.newaxis], y) + + # Evaluate the models using crossvalidation + scores = cross_val_score(pipeline, X[:, np.newaxis], y, + scoring="neg_mean_squared_error", cv=10) + + X_test = np.linspace(0, 1, 100) + plt.plot(X_test, pipeline.predict(X_test[:, np.newaxis]), label="Model") + plt.plot(X_test, true_fun(X_test), label="True function") + plt.scatter(X, y, edgecolor='b', s=20, label="Samples") + plt.xlabel("x") + plt.ylabel("y") + plt.xlim((0, 1)) + plt.ylim((-2, 2)) + plt.legend(loc="best") + plt.title("Degree {}\nMSE = {:.2e}(+/- {:.2e})".format( + degrees[i], -scores.mean(), scores.std())) +plt.show() + +# Common imports +import os +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +from sklearn.linear_model import LinearRegression, Ridge, Lasso +from sklearn.model_selection import train_test_split +from sklearn.utils import resample +from sklearn.metrics import mean_squared_error +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("EoS.csv"),'r') + +# Read the EoS data as csv file and organize the data into two arrays with density and energies +EoS = pd.read_csv(infile, names=('Density', 'Energy')) +EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce') +EoS = EoS.dropna() +Energies = EoS['Energy'] +Density = EoS['Density'] +# The design matrix now as function of various polytrops + +Maxpolydegree = 30 +X = np.zeros((len(Density),Maxpolydegree)) +X[:,0] = 1.0 +testerror = np.zeros(Maxpolydegree) +trainingerror = np.zeros(Maxpolydegree) +polynomial = np.zeros(Maxpolydegree) + +trials = 100 +for polydegree in range(1, Maxpolydegree): + polynomial[polydegree] = polydegree + for degree in range(polydegree): + X[:,degree] = Density**(degree/3.0) + +# loop over trials in order to estimate the expectation value of the MSE + testerror[polydegree] = 0.0 + trainingerror[polydegree] = 0.0 + for samples in range(trials): + x_train, x_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2) + model = LinearRegression(fit_intercept=True).fit(x_train, y_train) + ypred = model.predict(x_train) + ytilde = model.predict(x_test) + testerror[polydegree] += mean_squared_error(y_test, ytilde) + trainingerror[polydegree] += mean_squared_error(y_train, ypred) + + testerror[polydegree] /= trials + trainingerror[polydegree] /= trials + print("Degree of polynomial: %3d"% polynomial[polydegree]) + print("Mean squared error on training data: %.8f" % trainingerror[polydegree]) + print("Mean squared error on test data: %.8f" % testerror[polydegree]) + +plt.plot(polynomial, np.log10(trainingerror), label='Training Error') +plt.plot(polynomial, np.log10(testerror), label='Test Error') +plt.xlabel('Polynomial degree') +plt.ylabel('log10[MSE]') +plt.legend() +plt.show() + +# Common imports +import os +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +from sklearn.linear_model import LinearRegression, Ridge, Lasso +from sklearn.metrics import mean_squared_error +from sklearn.model_selection import KFold +from sklearn.model_selection import cross_val_score + + +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("EoS.csv"),'r') + +# Read the EoS data as csv file and organize the data into two arrays with density and energies +EoS = pd.read_csv(infile, names=('Density', 'Energy')) +EoS['Energy'] = pd.to_numeric(EoS['Energy'], errors='coerce') +EoS = EoS.dropna() +Energies = EoS['Energy'] +Density = EoS['Density'] +# The design matrix now as function of various polytrops + +Maxpolydegree = 30 +X = np.zeros((len(Density),Maxpolydegree)) +X[:,0] = 1.0 +estimated_mse_sklearn = np.zeros(Maxpolydegree) +polynomial = np.zeros(Maxpolydegree) +k =5 +kfold = KFold(n_splits = k) + +for polydegree in range(1, Maxpolydegree): + polynomial[polydegree] = polydegree + for degree in range(polydegree): + X[:,degree] = Density**(degree/3.0) + OLS = LinearRegression() +# loop over trials in order to estimate the expectation value of the MSE + estimated_mse_folds = cross_val_score(OLS, X, Energies, scoring='neg_mean_squared_error', cv=kfold) +#[:, np.newaxis] + estimated_mse_sklearn[polydegree] = np.mean(-estimated_mse_folds) + +plt.plot(polynomial, np.log10(estimated_mse_sklearn), label='Test Error') +plt.xlabel('Polynomial degree') +plt.ylabel('log10[MSE]') +plt.legend() +plt.show() + +import numpy as np +import matplotlib.pyplot as plt +from sklearn.model_selection import KFold +from sklearn.linear_model import Ridge +from sklearn.model_selection import cross_val_score +from sklearn.preprocessing import PolynomialFeatures + +# A seed just to ensure that the random numbers are the same for every run. +np.random.seed(3155) +# Generate the data. +n = 100 +x = np.linspace(-3, 3, n).reshape(-1, 1) +y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape) +# Decide degree on polynomial to fit +poly = PolynomialFeatures(degree = 10) + +# Decide which values of lambda to use +nlambdas = 500 +lambdas = np.logspace(-3, 5, nlambdas) +# Initialize a KFold instance +k = 5 +kfold = KFold(n_splits = k) +estimated_mse_sklearn = np.zeros(nlambdas) +i = 0 +for lmb in lambdas: + ridge = Ridge(alpha = lmb) + estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold) + estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds) + i += 1 +plt.figure() +plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score') +plt.xlabel('log10(lambda)') +plt.ylabel('MSE') +plt.legend() +plt.show() \ No newline at end of file diff --git a/doc/LectureNotes/_build/jupyter_execute/chapter2_25_2.png b/doc/LectureNotes/_build/jupyter_execute/chapter2_25_2.png new file mode 100644 index 0000000000000000000000000000000000000000..a95b9386fdc3380b7947f57167339be29ae3d9f9 GIT binary patch literal 4742 zcma)=2~-pJ+QvhIL|Ls?5Q0rDgIEzzKtR@OU2s5_3Ii2TabcB&$i9Vy;&QDbWhhXP zrC3BA3{r&<1Ys1nAOb=oJB)3Bf-I3G0kUv|_g;Jb+TQP;IhmQ0^Ui;A@_V1>eP%AW zyE>{V>nP)JI5lwpo`X1?Lah9Jex;Ipuk~fAqx`TlYVTK39^rvevBx3;a4yH9Xkp<| zVIjvqjR}Z|3<;;28Oyh=p9V)o(ITx)OepUQjKd>>OrGUb=gSvaMcaQU5{FY+yZlkW z<>jx(;XaxH_w4eF%b)0HRQLta2j7YeEY6)N&FH3f?KvOj*#sPy&$+6ZFeIfAcp47h&9l%>Drmy>@n1jy?R6g_O(MPfs z*@tc5ExuaEZHf!czh4lh(>-5}-M^PO)!&~u*C#p^7}Ofy64_#(E?m5uUSu~ChA4vv zkJg{4B7F+#I5ZxHtxs~C51vk@Iw6~A7I)xLx-nH6rlxxjYy|~^B zB8~L2H$9|_g%iy!E#)z0z-T_q=bmNnG|~e%nD*)4+<7hUqL0Ex~LCc^fB+t ztXng`3&Z^llDa^4D&u%{b|^9BRPVGgc3Ed0U$K!pyuI2Nj?(#+D-_%9?rCIRZOqcrJDUjgYcx0C z4H>bGR^c2AcyVw_a@-?tYxqLhj>l~|Pa1nQ_WwsOyn~utj6^;xaih#$^qYI~7p&fQ z&;Qg09~Py}nk9mTIGOUirr-|IiY_&bw(|=n zMx>He_N)MB?V2Tb=q#;o?d-N3%p~=4THODYH9+^ z55tw;W(@bGkZ&Ms!FnRclB?H9DnsC(4<6wlLV{Q!FfcH?@-{F!zE7vdn|DE&q|8x? zBM4?kN3(@V2;`s;M(UxTGRy>AxL02n@`u19rxUc}Yy>#k%C3H7LTQrv{5Y8fXgn2#_UB1tQOYX}1 zx@_z;b|=26L47>`dL5JWq}>{9zY}*B`Ij2g61J3-m~D+Zg%5ZBp$h#YL%tT0cjIoq z#^H9pRm5M*_d{;)HO{{uV6Hi^d!LW5Zq0u>$-8pB^NH{bED3?#*mKUAN!z(E0kU(% zE<;9$uR{+zyvV&*A5{6O>6vsN_PTiPN{WUT?-VnLX~#;^5F9XE@-H(0^R8w*wBveV z1?c&n1ZFj!91f}DJJ~~rDFtkLP?-ss2Y=ZbxwH-&VR~22U1SE8jh$sD>vpv-s}+Zg zodsH}7LStb{o(uQ5vH?o&B@#sIj;)Owk$AyDUEfl(f4W9+)!h%a18T=`Yl_u$`L8F z!viuSSyFXJiuFVKNdbIWAm#}B5$aUushFn!Zg{5mH@YG1Do@UqaM@rdIZj_N_fxeW zT=~OZ$4O{z2CAw;*n}x1dk<(yaek}d1gnRNO|p}2usYa`QY7>iD;p@_IkiMC;RI&j zWz+9H9NL;Ra9UNS#a)Me95MZ&xk(-ts6J%iIxa;RfsXcm^hHZ-+~2V zsnDDbrtR-m)BqN{=kpRUmA#ZzH5o%VP39l+}nf_sYy#(5*(CuTNp$}PBZ#rI1C05q{^ z?>2?zjIe2d#H?j@md~Co3w5X<`v(S=dL`JVb$$^<+iz{Tbu=s&Nzjx-+9U_tL00!o zSxZ?^tWf)b!jMxvZSDn|=gFyp7WlziNKmH42&1Uh15!Cu)G?(W?Wg4IrbuTf;vl3c zqo#6C;Zk|l`T(CmGuk$8yz0KiSTCMIrmbRl!-jqzyS5OwAM|0=r!Th% z&R2-28#Yl?szY?h9f#>Ek#(LH^7{$Lqd(B^{b- zvDgdLfl!~%mw5@pwa+XTfI^%bzO#xV!gb;tQ$|umm+wA+OCG01u$##Whx}(+;>WaT zjmmP*^HI3%o>iirW@@B2s*5d5=?p*nLlpb}SiQ?@9~Q+s#NLb?7@^Z6`0^|l!B1BA z2?jQS*LVKW)|XXGc8%D=CF~CPQ^suRLa7qv~Q-dE3JkEdAjll`pvDd!xa9xINCBov?dgM zvkM885M2fo`tbS0q#Fp;+E<#Jo10fqu;CQRwsG&+U;_DD{hHfI;Bc>nFhf3snMVl& zO1i}Q@A&SJo%v)Zez%`bYkuv~E}%4E0H@1Bo+`*2ylce)pih7FfU}zPwAz|hVX^6i z##Qi)SQG1NTRe@W1$jB$(dz2;YRKRlt_^<9@l#ErRFSp^3P<~1o@ns8=2jDb^?ZYs zrOACa!>|sKapf^yy*mq$qB35B^xA&Y+P1n^HBI&fHVNZ7^t?JSUpXG%Wl>pQl?B43 zC)d1|>H9mv&Y6C=I+B%nN@oUOm$Ha!G16!B}dxa z{cw&;)msyJ`pV3AnnvBb&w`UK)4_?!kf9swo2ly(0dTWabZ6!!2C?9>qL8`k_hny+ zzvJ6vN_N4HJ>r;GzQ+$V0Rv^lzv}APPgWlIZEEUkK%ALWbw6k+&PN(@+u?%!{p(W8 zXn=bi_Os>M-|D5gO-oTXQ_fYPDRw@yPHBF1#cf1pOw$sC4|m6^2yUn3B8$mNM|B$# zeJDjwCs3P~&Amk3a%2PgnypUTQSqal5MOOXYwley5$BgD4!<>+5XK*gyhdLfT7Cdp zAs7a{(`#SqaSGmwE&>k9NhkW!HoTm;)Zn(HZUsSu#5(o)@DtRTm4f;Um6A%tCr1N2jHVRNE#V}(0^LE`)p(Mi%~o7kuPpjPDM**TCQiIX6l5bn z+ZMfG@*f;Gsd2lT_DejzjwjmBTPXvvfrVG#=st+Z*ltd3x9uQHs zS?n!jEowqwsOKvdmLYC#*LXknzBd!w!HJFR{GX<>y^`71WykdCiwrmnC;KS%Wv5R? zd$RO1K@oGr;MK!xzA;5TVuIizCX1l;d)9rMT_>iAQT${Wus9(^-(+4TSrB=yonJ1EECP^ zp)4MiTAjVM>L_STl*K^(JAT^7YBL=uo6+{`mN_6}PS}j9lE&3mJL20bw|zNnG1bO?%^Si7(m9V|NX z+;KVzeu0}<@H8Y330E8VEOd|EI+ry&S}8?;gysx0L;0q+N{)u(17X_-{-MM2;YGd5 z+VH|@aW-0MNo)E7TULu6YB*2`%dd%GA_=$X+xoa?%uXDZ2llCu)*PIJvZTemI~KQ8 z_@J*VSEkEB3Q|7uHxLM-eC+P*Mkm>IBu?TD$)Zqg2~sWn&Q*P?>(!De7h*9wH6|MJ zvil8~K!al4nz*Hpd+vQ~7=N2uhIo7PVld)*RrdkYg1{o4)1oIjMFH24YjnNnm^I(}&#r-G-JO7@7q^guG$+;)1B>{A#*PgI@fbu6im zgchy8RQk;WhP$ZxsoSjJh*}mUXkEQ~LEGS<2+w%tzxc~G8c3us-V};r&pbHaxi@n> xwVM2Gi8|{|h&DKuwfw5&!_oVGOw2MAGW8lQ$%mPF@^?ZwaIfp0@-O_q{s-fPUl{-Z literal 0 HcmV?d00001 diff --git a/doc/LectureNotes/_build/jupyter_execute/chapter3.ipynb b/doc/LectureNotes/_build/jupyter_execute/chapter3.ipynb new file mode 100644 index 000000000..ce3ab0adf --- /dev/null +++ b/doc/LectureNotes/_build/jupyter_execute/chapter3.ipynb @@ -0,0 +1,1508 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Ridge and Lasso Regression\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember10.mp4?vrtx=view-as-webpage)\n", + "\n", + "\n", + "## The singular value decomposition\n", + "\n", + "The examples we have looked at so far are cases where we normally can\n", + "invert the matrix $\\boldsymbol{X}^T\\boldsymbol{X}$. Using a polynomial expansion as we\n", + "did both for the masses and the fitting of the equation of state,\n", + "leads to row vectors of the design matrix which are essentially\n", + "orthogonal due to the polynomial character of our model. Obtaining the inverse of the design matrix is then often done via a so-called LU, QR or Cholesky decomposition. \n", + "\n", + "\n", + "\n", + "This may\n", + "however not the be case in general and a standard matrix inversion\n", + "algorithm based on say LU, QR or Cholesky decomposition may lead to singularities. We will see examples of this below.\n", + "\n", + "There is however a way to partially circumvent this problem and also gain some insights about the ordinary least squares approach, and later shrinkage methods like Ridge and Lasso regressions. \n", + "\n", + "This is given by the **Singular Value Decomposition** algorithm, perhaps\n", + "the most powerful linear algebra algorithm. Let us look at a\n", + "different example where we may have problems with the standard matrix\n", + "inversion algorithm. Thereafter we dive into the math of the SVD.\n", + "\n", + "\n", + "\n", + "One of the typical problems we encounter with linear regression, in particular \n", + "when the matrix $\\boldsymbol{X}$ (our so-called design matrix) is high-dimensional, \n", + "are problems with near singular or singular matrices. The column vectors of $\\boldsymbol{X}$ \n", + "may be linearly dependent, normally referred to as super-collinearity. \n", + "This means that the matrix may be rank deficient and it is basically impossible to \n", + "to model the data using linear regression. As an example, consider the matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "\\mathbf{X} & = \\left[\n", + "\\begin{array}{rrr}\n", + "1 & -1 & 2\n", + "\\\\\n", + "1 & 0 & 1\n", + "\\\\\n", + "1 & 2 & -1\n", + "\\\\\n", + "1 & 1 & 0\n", + "\\end{array} \\right]\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The columns of $\\boldsymbol{X}$ are linearly dependent. We see this easily since the \n", + "the first column is the row-wise sum of the other two columns. The rank (more correct,\n", + "the column rank) of a matrix is the dimension of the space spanned by the\n", + "column vectors. Hence, the rank of $\\mathbf{X}$ is equal to the number\n", + "of linearly independent columns. In this particular case the matrix has rank 2.\n", + "\n", + "Super-collinearity of an $(n \\times p)$-dimensional design matrix $\\mathbf{X}$ implies\n", + "that the inverse of the matrix $\\boldsymbol{X}^T\\boldsymbol{X}$ (the matrix we need to invert to solve the linear regression equations) is non-invertible. If we have a square matrix that does not have an inverse, we say this matrix singular. The example here demonstrates this" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "\\boldsymbol{X} & = \\left[\n", + "\\begin{array}{rr}\n", + "1 & -1\n", + "\\\\\n", + "1 & -1\n", + "\\end{array} \\right].\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We see easily that $\\mbox{det}(\\boldsymbol{X}) = x_{11} x_{22} - x_{12} x_{21} = 1 \\times (-1) - 1 \\times (-1) = 0$. Hence, $\\mathbf{X}$ is singular and its inverse is undefined.\n", + "This is equivalent to saying that the matrix $\\boldsymbol{X}$ has at least an eigenvalue which is zero.\n", + "\n", + "\n", + "If our design matrix $\\boldsymbol{X}$ which enters the linear regression problem" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "
\n", + "\n", + "$$\n", + "\\begin{equation}\n", + "\\boldsymbol{\\beta} = (\\boldsymbol{X}^{T} \\boldsymbol{X})^{-1} \\boldsymbol{X}^{T} \\boldsymbol{y},\n", + "\\label{_auto1} \\tag{1}\n", + "\\end{equation}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "has linearly dependent column vectors, we will not be able to compute the inverse\n", + "of $\\boldsymbol{X}^T\\boldsymbol{X}$ and we cannot find the parameters (estimators) $\\beta_i$. \n", + "The estimators are only well-defined if $(\\boldsymbol{X}^{T}\\boldsymbol{X})^{-1}$ exits. \n", + "This is more likely to happen when the matrix $\\boldsymbol{X}$ is high-dimensional. In this case it is likely to encounter a situation where \n", + "the regression parameters $\\beta_i$ cannot be estimated.\n", + "\n", + "A cheap *ad hoc* approach is simply to add a small diagonal component to the matrix to invert, that is we change" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^{T} \\boldsymbol{X} \\rightarrow \\boldsymbol{X}^{T} \\boldsymbol{X}+\\lambda \\boldsymbol{I},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\boldsymbol{I}$ is the identity matrix. When we discuss **Ridge** regression this is actually what we end up evaluating. The parameter $\\lambda$ is called a hyperparameter. More about this later. \n", + "\n", + "\n", + "\n", + "\n", + "\n", + "From standard linear algebra we know that a square matrix $\\boldsymbol{X}$ can be diagonalized if and only it is \n", + "a so-called [normal matrix](https://en.wikipedia.org/wiki/Normal_matrix), that is if $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times n}$\n", + "we have $\\boldsymbol{X}\\boldsymbol{X}^T=\\boldsymbol{X}^T\\boldsymbol{X}$ or if $\\boldsymbol{X}\\in {\\mathbb{C}}^{n\\times n}$ we have $\\boldsymbol{X}\\boldsymbol{X}^{\\dagger}=\\boldsymbol{X}^{\\dagger}\\boldsymbol{X}$.\n", + "The matrix has then a set of eigenpairs" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\lambda_1,\\boldsymbol{u}_1),\\dots, (\\lambda_n,\\boldsymbol{u}_n),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and the eigenvalues are given by the diagonal matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\Sigma}=\\mathrm{Diag}(\\lambda_1, \\dots,\\lambda_n).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The matrix $\\boldsymbol{X}$ can be written in terms of an orthogonal/unitary transformation $\\boldsymbol{U}$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{U}\\boldsymbol{U}^T=\\boldsymbol{I}$ or $\\boldsymbol{U}\\boldsymbol{U}^{\\dagger}=\\boldsymbol{I}$.\n", + "\n", + "Not all square matrices are diagonalizable. A matrix like the one discussed above" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\begin{bmatrix} \n", + "1& -1 \\\\\n", + "1& -1\\\\\n", + "\\end{bmatrix}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "is not diagonalizable, it is a so-called [defective matrix](https://en.wikipedia.org/wiki/Defective_matrix). It is easy to see that the condition\n", + "$\\boldsymbol{X}\\boldsymbol{X}^T=\\boldsymbol{X}^T\\boldsymbol{X}$ is not fulfilled. \n", + "\n", + "\n", + "\n", + "## The SVD, a Fantastic Algorithm\n", + "\n", + "\n", + "However, and this is the strength of the SVD algorithm, any general\n", + "matrix $\\boldsymbol{X}$ can be decomposed in terms of a diagonal matrix and\n", + "two orthogonal/unitary matrices. The [Singular Value Decompostion\n", + "(SVD) theorem](https://en.wikipedia.org/wiki/Singular_value_decomposition)\n", + "states that a general $m\\times n$ matrix $\\boldsymbol{X}$ can be written in\n", + "terms of a diagonal matrix $\\boldsymbol{\\Sigma}$ of dimensionality $m\\times n$\n", + "and two orthognal matrices $\\boldsymbol{U}$ and $\\boldsymbol{V}$, where the first has\n", + "dimensionality $m \\times m$ and the last dimensionality $n\\times n$.\n", + "We have then" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "As an example, the above defective matrix can be decomposed as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\frac{1}{\\sqrt{2}}\\begin{bmatrix} 1& 1 \\\\ 1& -1\\\\ \\end{bmatrix} \\begin{bmatrix} 2& 0 \\\\ 0& 0\\\\ \\end{bmatrix} \\frac{1}{\\sqrt{2}}\\begin{bmatrix} 1& -1 \\\\ 1& 1\\\\ \\end{bmatrix}=\\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with eigenvalues $\\sigma_1=2$ and $\\sigma_2=0$. \n", + "The SVD exits always! \n", + "\n", + "The SVD\n", + "decomposition (singular values) gives eigenvalues \n", + "$\\sigma_i\\geq\\sigma_{i+1}$ for all $i$ and for dimensions larger than $i=p$, the\n", + "eigenvalues (singular values) are zero.\n", + "\n", + "In the general case, where our design matrix $\\boldsymbol{X}$ has dimension\n", + "$n\\times p$, the matrix is thus decomposed into an $n\\times n$\n", + "orthogonal matrix $\\boldsymbol{U}$, a $p\\times p$ orthogonal matrix $\\boldsymbol{V}$\n", + "and a diagonal matrix $\\boldsymbol{\\Sigma}$ with $r=\\mathrm{min}(n,p)$\n", + "singular values $\\sigma_i\\geq 0$ on the main diagonal and zeros filling\n", + "the rest of the matrix. There are at most $p$ singular values\n", + "assuming that $n > p$. In our regression examples for the nuclear\n", + "masses and the equation of state this is indeed the case, while for\n", + "the Ising model we have $p > n$. These are often cases that lead to\n", + "near singular or singular matrices.\n", + "\n", + "The columns of $\\boldsymbol{U}$ are called the left singular vectors while the columns of $\\boldsymbol{V}$ are the right singular vectors.\n", + "\n", + "## Economy-size SVD\n", + "\n", + "If we assume that $n > p$, then our matrix $\\boldsymbol{U}$ has dimension $n\n", + "\\times n$. The last $n-p$ columns of $\\boldsymbol{U}$ become however\n", + "irrelevant in our calculations since they are multiplied with the\n", + "zeros in $\\boldsymbol{\\Sigma}$.\n", + "\n", + "The economy-size decomposition removes extra rows or columns of zeros\n", + "from the diagonal matrix of singular values, $\\boldsymbol{\\Sigma}$, along with the columns\n", + "in either $\\boldsymbol{U}$ or $\\boldsymbol{V}$ that multiply those zeros in the expression. \n", + "Removing these zeros and columns can improve execution time\n", + "and reduce storage requirements without compromising the accuracy of\n", + "the decomposition.\n", + "\n", + "If $n > p$, we keep only the first $p$ columns of $\\boldsymbol{U}$ and $\\boldsymbol{\\Sigma}$ has dimension $p\\times p$. \n", + "If $p > n$, then only the first $n$ columns of $\\boldsymbol{V}$ are computed and $\\boldsymbol{\\Sigma}$ has dimension $n\\times n$.\n", + "The $n=p$ case is obvious, we retain the full SVD. \n", + "In general the economy-size SVD leads to less FLOPS and still conserving the desired accuracy." + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[[ 1. -1. 2.]\n", + " [ 1. 0. 1.]\n", + " [ 1. 2. -1.]\n", + " [ 1. 1. 0.]]\n", + "[[ 4. 2. 2.]\n", + " [ 2. 6. -4.]\n", + " [ 2. -4. 6.]]\n", + "[[-1.18404906e-16 8.16496581e-01 -5.77350269e-01]\n", + " [-7.07106781e-01 4.08248290e-01 5.77350269e-01]\n", + " [ 7.07106781e-01 4.08248290e-01 5.77350269e-01]]\n", + "[1.00000000e+01 6.00000000e+00 9.10898112e-32]\n", + "[[ 3.33066907e-17 -7.07106781e-01 7.07106781e-01]\n", + " [ 8.16496581e-01 4.08248290e-01 4.08248290e-01]\n", + " [ 5.77350269e-01 -5.77350269e-01 -5.77350269e-01]]\n", + "[[-3.65939208e+30 3.65939208e+30 3.65939208e+30]\n", + " [ 3.65939208e+30 -3.65939208e+30 -3.65939208e+30]\n", + " [ 3.65939208e+30 -3.65939208e+30 -3.65939208e+30]]\n" + ] + } + ], + "source": [ + "import numpy as np\n", + "# SVD inversion\n", + "def SVDinv(A):\n", + " ''' Takes as input a numpy matrix A and returns inv(A) based on singular value decomposition (SVD).\n", + " SVD is numerically more stable than the inversion algorithms provided by\n", + " numpy and scipy.linalg at the cost of being slower.\n", + " '''\n", + " U, s, VT = np.linalg.svd(A)\n", + "# print('test U')\n", + "# print( (np.transpose(U) @ U - U @np.transpose(U)))\n", + "# print('test VT')\n", + "# print( (np.transpose(VT) @ VT - VT @np.transpose(VT)))\n", + " print(U)\n", + " print(s)\n", + " print(VT)\n", + "\n", + " D = np.zeros((len(U),len(VT)))\n", + " for i in range(0,len(VT)):\n", + " D[i,i]=s[i]\n", + " UT = np.transpose(U); V = np.transpose(VT); invD = np.linalg.inv(D)\n", + " return np.matmul(V,np.matmul(invD,UT))\n", + "\n", + "\n", + "X = np.array([ [1.0, -1.0, 2.0], [1.0, 0.0, 1.0], [1.0, 2.0, -1.0], [1.0, 1.0, 0.0] ])\n", + "print(X)\n", + "A = np.transpose(X) @ X\n", + "print(A)\n", + "# Brute force inversion of super-collinear matrix\n", + "#B = np.linalg.inv(A)\n", + "#print(B)\n", + "C = SVDinv(A)\n", + "print(C)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The matrix $\\boldsymbol{X}$ has columns that are linearly dependent. The first\n", + "column is the row-wise sum of the other two columns. The rank of a\n", + "matrix (the column rank) is the dimension of space spanned by the\n", + "column vectors. The rank of the matrix is the number of linearly\n", + "independent columns, in this case just $2$. We see this from the\n", + "singular values when running the above code. Running the standard\n", + "inversion algorithm for matrix inversion with $\\boldsymbol{X}^T\\boldsymbol{X}$ results\n", + "in the program terminating due to a singular matrix.\n", + "\n", + "\n", + "\n", + "\n", + "There are several interesting mathematical properties which will be\n", + "relevant when we are going to discuss the differences between say\n", + "ordinary least squares (OLS) and **Ridge** regression.\n", + "\n", + "We have from OLS that the parameters of the linear approximation are given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\beta} = \\boldsymbol{X}\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The matrix to invert can be rewritten in terms of our SVD decomposition as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{X} = \\boldsymbol{V}\\boldsymbol{\\Sigma}^T\\boldsymbol{U}^T\\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Using the orthogonality properties of $\\boldsymbol{U}$ we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{X} = \\boldsymbol{V}\\boldsymbol{\\Sigma}^T\\boldsymbol{\\Sigma}\\boldsymbol{V}^T = \\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{D}$ being a diagonal matrix with values along the diagonal given by the singular values squared. \n", + "\n", + "This means that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{X}^T\\boldsymbol{X})\\boldsymbol{V} = \\boldsymbol{V}\\boldsymbol{D},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "that is the eigenvectors of $(\\boldsymbol{X}^T\\boldsymbol{X})$ are given by the columns of the right singular matrix of $\\boldsymbol{X}$ and the eigenvalues are the squared singular values. It is easy to show (show this) that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{X}\\boldsymbol{X}^T)\\boldsymbol{U} = \\boldsymbol{U}\\boldsymbol{D},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "that is, the eigenvectors of $(\\boldsymbol{X}\\boldsymbol{X})^T$ are the columns of the left singular matrix and the eigenvalues are the same. \n", + "\n", + "Going back to our OLS equation we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}\\boldsymbol{\\beta} = \\boldsymbol{X}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T \\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}=\\boldsymbol{U\\Sigma V^T}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T \\right)^{-1}(\\boldsymbol{U\\Sigma V^T})^T\\boldsymbol{y}=\\boldsymbol{U}\\boldsymbol{U}^T\\boldsymbol{y}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We will come back to this expression when we discuss Ridge regression. \n", + "\n", + "\n", + "$$ \\tilde{y}^{OLS}=\\boldsymbol{X}\\hat{\\beta}^{OLS}=\\sum_{j=1}^p \\boldsymbol{u}_j\\boldsymbol{u}_j^T\\boldsymbol{y}$$ and for Ridge we have \n", + "\n", + "$$ \\tilde{y}^{Ridge}=\\boldsymbol{X}\\hat{\\beta}^{Ridge}=\\sum_{j=1}^p \\boldsymbol{u}_j\\frac{\\sigma_j^2}{\\sigma_j^2+\\lambda}\\boldsymbol{u}_j^T\\boldsymbol{y}$$ . \n", + "\n", + "It is indeed the economy-sized SVD, note the summation runs up tp $$p$$ only and not $$n$$. \n", + "\n", + "Here we have that $$\\boldsymbol{X} = \\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T$$, with $$\\Sigma$$ being an $$ n\\times p$$ matrix and $$\\boldsymbol{V}$$ being a $$ p\\times p$$ matrix. We also have assumed here that $$ n > p$$. \n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Ridge and LASSO Regression\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember11.mp4?vrtx=view-as-webpage)\n", + "\n", + "Let us remind ourselves about the expression for the standard Mean Squared Error (MSE) which we used to define our cost function and the equations for the ordinary least squares (OLS) method, that is \n", + "our optimization problem is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in {\\mathbb{R}}^{p}}}\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)\\right\\}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or we can state it as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in\n", + "{\\mathbb{R}}^{p}}}\\frac{1}{n}\\sum_{i=0}^{n-1}\\left(y_i-\\tilde{y}_i\\right)^2=\\frac{1}{n}\\vert\\vert \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\vert\\vert_2^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have used the definition of a norm-2 vector, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\vert\\vert \\boldsymbol{x}\\vert\\vert_2 = \\sqrt{\\sum_i x_i^2}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "By minimizing the above equation with respect to the parameters\n", + "$\\boldsymbol{\\beta}$ we could then obtain an analytical expression for the\n", + "parameters $\\boldsymbol{\\beta}$. We can add a regularization parameter $\\lambda$ by\n", + "defining a new cost function to be optimized, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in\n", + "{\\mathbb{R}}^{p}}}\\frac{1}{n}\\vert\\vert \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\vert\\vert_2^2+\\lambda\\vert\\vert \\boldsymbol{\\beta}\\vert\\vert_2^2\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which leads to the Ridge regression minimization problem where we\n", + "require that $\\vert\\vert \\boldsymbol{\\beta}\\vert\\vert_2^2\\le t$, where $t$ is\n", + "a finite number larger than zero. By defining" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{X},\\boldsymbol{\\beta})=\\frac{1}{n}\\vert\\vert \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\vert\\vert_2^2+\\lambda\\vert\\vert \\boldsymbol{\\beta}\\vert\\vert_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we have a new optimization equation" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in\n", + "{\\mathbb{R}}^{p}}}\\frac{1}{n}\\vert\\vert \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\vert\\vert_2^2+\\lambda\\vert\\vert \\boldsymbol{\\beta}\\vert\\vert_1\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which leads to Lasso regression. Lasso stands for least absolute shrinkage and selection operator. \n", + "\n", + "Here we have defined the norm-1 as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\vert\\vert \\boldsymbol{x}\\vert\\vert_1 = \\sum_i \\vert x_i\\vert.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Using the matrix-vector expression for Ridge regression," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{X},\\boldsymbol{\\beta})=\\frac{1}{n}\\left\\{(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta})^T(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta})\\right\\}+\\lambda\\boldsymbol{\\beta}^T\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "by taking the derivatives with respect to $\\boldsymbol{\\beta}$ we obtain then\n", + "a slightly modified matrix inversion problem which for finite values\n", + "of $\\lambda$ does not suffer from singularity problems. We obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{Ridge}} = \\left(\\boldsymbol{X}^T\\boldsymbol{X}+\\lambda\\boldsymbol{I}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{I}$ being a $p\\times p$ identity matrix with the constraint that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\sum_{i=0}^{p-1} \\beta_i^2 \\leq t,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $t$ a finite positive number. \n", + "\n", + "We see that Ridge regression is nothing but the standard\n", + "OLS with a modified diagonal term added to $\\boldsymbol{X}^T\\boldsymbol{X}$. The\n", + "consequences, in particular for our discussion of the bias-variance tradeoff \n", + "are rather interesting.\n", + "\n", + "Furthermore, if we use the result above in terms of the SVD decomposition (our analysis was done for the OLS method), we had" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{X}\\boldsymbol{X}^T)\\boldsymbol{U} = \\boldsymbol{U}\\boldsymbol{D}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can analyse the OLS solutions in terms of the eigenvectors (the columns) of the right singular value matrix $\\boldsymbol{U}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}\\boldsymbol{\\beta} = \\boldsymbol{X}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T \\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}=\\boldsymbol{U\\Sigma V^T}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T \\right)^{-1}(\\boldsymbol{U\\Sigma V^T})^T\\boldsymbol{y}=\\boldsymbol{U}\\boldsymbol{U}^T\\boldsymbol{y}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "For Ridge regression this becomes" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}\\boldsymbol{\\beta}^{\\mathrm{Ridge}} = \\boldsymbol{U\\Sigma V^T}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T+\\lambda\\boldsymbol{I} \\right)^{-1}(\\boldsymbol{U\\Sigma V^T})^T\\boldsymbol{y}=\\sum_{j=0}^{p-1}\\boldsymbol{u}_j\\boldsymbol{u}_j^T\\frac{\\sigma_j^2}{\\sigma_j^2+\\lambda}\\boldsymbol{y},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with the vectors $\\boldsymbol{u}_j$ being the columns of $\\boldsymbol{U}$. \n", + "\n", + "\n", + "Since $\\lambda \\geq 0$, it means that compared to OLS, we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\sigma_j^2}{\\sigma_j^2+\\lambda} \\leq 1.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Ridge regression finds the coordinates of $\\boldsymbol{y}$ with respect to the\n", + "orthonormal basis $\\boldsymbol{U}$, it then shrinks the coordinates by\n", + "$\\frac{\\sigma_j^2}{\\sigma_j^2+\\lambda}$. Recall that the SVD has\n", + "eigenvalues ordered in a descending way, that is $\\sigma_i \\geq\n", + "\\sigma_{i+1}$.\n", + "\n", + "For small eigenvalues $\\sigma_i$ it means that their contributions become less important, a fact which can be used to reduce the number of degrees of freedom.\n", + "Actually, calculating the variance of $\\boldsymbol{X}\\boldsymbol{v}_j$ shows that this quantity is equal to $\\sigma_j^2/n$.\n", + "With a parameter $\\lambda$ we can thus shrink the role of specific parameters. \n", + "\n", + "\n", + "\n", + "For the sake of simplicity, let us assume that the design matrix is orthonormal, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{X}=(\\boldsymbol{X}^T\\boldsymbol{X})^{-1} =\\boldsymbol{I}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In this case the standard OLS results in" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{OLS}} = \\boldsymbol{X}^T\\boldsymbol{y}=\\sum_{i=0}^{p-1}\\boldsymbol{u}_j\\boldsymbol{u}_j^T\\boldsymbol{y},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{Ridge}} = \\left(\\boldsymbol{I}+\\lambda\\boldsymbol{I}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}=\\left(1+\\lambda\\right)^{-1}\\boldsymbol{\\beta}^{\\mathrm{OLS}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "that is the Ridge estimator scales the OLS estimator by the inverse of a factor $1+\\lambda$, and\n", + "the Ridge estimator converges to zero when the hyperparameter goes to\n", + "infinity.\n", + "\n", + "We will come back to more interpreations after we have gone through some of the statistical analysis part. \n", + "\n", + "For more discussions of Ridge and Lasso regression, [Wessel van Wieringen's](https://arxiv.org/abs/1509.09169) article is highly recommended.\n", + "Similarly, [Mehta et al's article](https://arxiv.org/abs/1803.08823) is also recommended.\n", + "\n", + "\n", + "\n", + "## A better understanding of regularization\n", + "\n", + "The parameter $\\lambda$ that we have introduced in the Ridge (and\n", + "Lasso as well) regression is often called a regularization parameter\n", + "or shrinkage parameter. It is common to call it a hyperparameter. What does it mean mathemtically?\n", + "\n", + "Here we will first look at how to analyze the difference between the\n", + "standard OLS equations and the Ridge expressions in terms of a linear\n", + "algebra analysis using the SVD algorithm. Thereafter, we will link\n", + "(see the material on the bias-variance tradeoff below) these\n", + "observation to the statisical analysis of the results. In particular\n", + "we consider how the variance of the parameters $\\boldsymbol{\\beta}$ is\n", + "affected by changing the parameter $\\lambda$.\n", + "\n", + "\n", + "We have our design matrix\n", + " $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$. With the SVD we decompose it as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\boldsymbol{U\\Sigma V^T},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{U}\\in {\\mathbb{R}}^{n\\times n}$, $\\boldsymbol{\\Sigma}\\in {\\mathbb{R}}^{n\\times p}$\n", + "and $\\boldsymbol{V}\\in {\\mathbb{R}}^{p\\times p}$.\n", + "\n", + "The matrices $\\boldsymbol{U}$ and $\\boldsymbol{V}$ are unitary/orthonormal matrices, that is in case the matrices are real we have $\\boldsymbol{U}^T\\boldsymbol{U}=\\boldsymbol{U}\\boldsymbol{U}^T=\\boldsymbol{I}$ and $\\boldsymbol{V}^T\\boldsymbol{V}=\\boldsymbol{V}\\boldsymbol{V}^T=\\boldsymbol{I}$.\n", + "\n", + "\n", + "\n", + "## Introducing the Covariance and Correlation functions\n", + "\n", + "Before we discuss the link between for example Ridge regression and the singular value decomposition, we need to remind ourselves about\n", + "the definition of the covariance and the correlation function. These are quantities \n", + "\n", + "Suppose we have defined two vectors\n", + "$\\hat{x}$ and $\\hat{y}$ with $n$ elements each. The covariance matrix $\\boldsymbol{C}$ is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x},\\boldsymbol{y}] = \\begin{bmatrix} \\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{x}] & \\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}] \\\\\n", + " \\mathrm{cov}[\\boldsymbol{y},\\boldsymbol{x}] & \\mathrm{cov}[\\boldsymbol{y},\\boldsymbol{y}] \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where for example" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}] =\\frac{1}{n} \\sum_{i=0}^{n-1}(x_i- \\overline{x})(y_i- \\overline{y}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With this definition and recalling that the variance is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathrm{var}[\\boldsymbol{x}]=\\frac{1}{n} \\sum_{i=0}^{n-1}(x_i- \\overline{x})^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we can rewrite the covariance matrix as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x},\\boldsymbol{y}] = \\begin{bmatrix} \\mathrm{var}[\\boldsymbol{x}] & \\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}] \\\\\n", + " \\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}] & \\mathrm{var}[\\boldsymbol{y}] \\\\\n", + " \\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The covariance takes values between zero and infinity and may thus\n", + "lead to problems with loss of numerical precision for particularly\n", + "large values. It is common to scale the covariance matrix by\n", + "introducing instead the correlation matrix defined via the so-called\n", + "correlation function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathrm{corr}[\\boldsymbol{x},\\boldsymbol{y}]=\\frac{\\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}]}{\\sqrt{\\mathrm{var}[\\boldsymbol{x}] \\mathrm{var}[\\boldsymbol{y}]}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The correlation function is then given by values $\\mathrm{corr}[\\boldsymbol{x},\\boldsymbol{y}]\n", + "\\in [-1,1]$. This avoids eventual problems with too large values. We\n", + "can then define the correlation matrix for the two vectors $\\boldsymbol{x}$\n", + "and $\\boldsymbol{y}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{K}[\\boldsymbol{x},\\boldsymbol{y}] = \\begin{bmatrix} 1 & \\mathrm{corr}[\\boldsymbol{x},\\boldsymbol{y}] \\\\\n", + " \\mathrm{corr}[\\boldsymbol{y},\\boldsymbol{x}] & 1 \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the above example this is the function we constructed using **pandas**.\n", + "\n", + "\n", + "\n", + "In our derivation of the various regression algorithms like **Ordinary Least Squares** or **Ridge regression**\n", + "we defined the design/feature matrix $\\boldsymbol{X}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\\begin{bmatrix}\n", + "x_{0,0} & x_{0,1} & x_{0,2}& \\dots & \\dots x_{0,p-1}\\\\\n", + "x_{1,0} & x_{1,1} & x_{1,2}& \\dots & \\dots x_{1,p-1}\\\\\n", + "x_{2,0} & x_{2,1} & x_{2,2}& \\dots & \\dots x_{2,p-1}\\\\\n", + "\\dots & \\dots & \\dots & \\dots \\dots & \\dots \\\\\n", + "x_{n-2,0} & x_{n-2,1} & x_{n-2,2}& \\dots & \\dots x_{n-2,p-1}\\\\\n", + "x_{n-1,0} & x_{n-1,1} & x_{n-1,2}& \\dots & \\dots x_{n-1,p-1}\\\\\n", + "\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$, with the predictors/features $p$ refering to the column numbers and the\n", + "entries $n$ being the row elements.\n", + "We can rewrite the design/feature matrix in terms of its column vectors as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\\begin{bmatrix} \\boldsymbol{x}_0 & \\boldsymbol{x}_1 & \\boldsymbol{x}_2 & \\dots & \\dots & \\boldsymbol{x}_{p-1}\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with a given vector" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_i^T = \\begin{bmatrix}x_{0,i} & x_{1,i} & x_{2,i}& \\dots & \\dots x_{n-1,i}\\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With these definitions, we can now rewrite our $2\\times 2$\n", + "correaltion/covariance matrix in terms of a moe general design/feature\n", + "matrix $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$. This leads to a $p\\times p$\n", + "covariance matrix for the vectors $\\boldsymbol{x}_i$ with $i=0,1,\\dots,p-1$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x}] = \\begin{bmatrix}\n", + "\\mathrm{var}[\\boldsymbol{x}_0] & \\mathrm{cov}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] & \\mathrm{cov}[\\boldsymbol{x}_0,\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{cov}[\\boldsymbol{x}_0,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\mathrm{cov}[\\boldsymbol{x}_1,\\boldsymbol{x}_0] & \\mathrm{var}[\\boldsymbol{x}_1] & \\mathrm{cov}[\\boldsymbol{x}_1,\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{cov}[\\boldsymbol{x}_1,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\mathrm{cov}[\\boldsymbol{x}_2,\\boldsymbol{x}_0] & \\mathrm{cov}[\\boldsymbol{x}_2,\\boldsymbol{x}_1] & \\mathrm{var}[\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{cov}[\\boldsymbol{x}_2,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "\\mathrm{cov}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_0] & \\mathrm{cov}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_1] & \\mathrm{cov}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_{2}] & \\dots & \\dots & \\mathrm{var}[\\boldsymbol{x}_{p-1}]\\\\\n", + "\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and the correlation matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{K}[\\boldsymbol{x}] = \\begin{bmatrix}\n", + "1 & \\mathrm{corr}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] & \\mathrm{corr}[\\boldsymbol{x}_0,\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{corr}[\\boldsymbol{x}_0,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\mathrm{corr}[\\boldsymbol{x}_1,\\boldsymbol{x}_0] & 1 & \\mathrm{corr}[\\boldsymbol{x}_1,\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{corr}[\\boldsymbol{x}_1,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\mathrm{corr}[\\boldsymbol{x}_2,\\boldsymbol{x}_0] & \\mathrm{corr}[\\boldsymbol{x}_2,\\boldsymbol{x}_1] & 1 & \\dots & \\dots & \\mathrm{corr}[\\boldsymbol{x}_2,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "\\mathrm{corr}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_0] & \\mathrm{corr}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_1] & \\mathrm{corr}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_{2}] & \\dots & \\dots & 1\\\\\n", + "\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The Numpy function **np.cov** calculates the covariance elements using\n", + "the factor $1/(n-1)$ instead of $1/n$ since it assumes we do not have\n", + "the exact mean values. The following simple function uses the\n", + "**np.vstack** function which takes each vector of dimension $1\\times n$\n", + "and produces a $2\\times n$ matrix $\\boldsymbol{W}$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{W} = \\begin{bmatrix} x_0 & y_0 \\\\\n", + " x_1 & y_1 \\\\\n", + " x_2 & y_2\\\\\n", + " \\dots & \\dots \\\\\n", + " x_{n-2} & y_{n-2}\\\\\n", + " x_{n-1} & y_{n-1} & \n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which in turn is converted into into the $2\\times 2$ covariance matrix\n", + "$\\boldsymbol{C}$ via the Numpy function **np.cov()**. We note that we can also calculate\n", + "the mean value of each set of samples $\\boldsymbol{x}$ etc using the Numpy\n", + "function **np.mean(x)**. We can also extract the eigenvalues of the\n", + "covariance matrix through the **np.linalg.eig()** function." + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "-0.03494744740413562\n", + "3.896222705525891\n", + "[[ 1.15493691 3.39431833]\n", + " [ 3.39431833 10.95648659]]\n" + ] + } + ], + "source": [ + "# Importing various packages\n", + "import numpy as np\n", + "n = 100\n", + "x = np.random.normal(size=n)\n", + "print(np.mean(x))\n", + "y = 4+3*x+np.random.normal(size=n)\n", + "print(np.mean(y))\n", + "W = np.vstack((x, y))\n", + "C = np.cov(W)\n", + "print(C)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The previous example can be converted into the correlation matrix by\n", + "simply scaling the matrix elements with the variances. We should also\n", + "subtract the mean values for each column. This leads to the following\n", + "code which sets up the correlations matrix for the previous example in\n", + "a more brute force way. Here we scale the mean values for each column of the design matrix, calculate the relevant mean values and variances and then finally set up the $2\\times 2$ correlation matrix (since we have only two vectors)." + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "0.0684659365902349\n", + "1.3864659735909566\n", + "[[1. 0.56083552]\n", + " [0.56083552 1. ]]\n" + ] + } + ], + "source": [ + "import numpy as np\n", + "n = 100\n", + "# define two vectors \n", + "x = np.random.random(size=n)\n", + "y = 4+3*x+np.random.normal(size=n)\n", + "#scaling the x and y vectors \n", + "x = x - np.mean(x)\n", + "y = y - np.mean(y)\n", + "variance_x = np.sum(x@x)/n\n", + "variance_y = np.sum(y@y)/n\n", + "print(variance_x)\n", + "print(variance_y)\n", + "cov_xy = np.sum(x@y)/n\n", + "cov_xx = np.sum(x@x)/n\n", + "cov_yy = np.sum(y@y)/n\n", + "C = np.zeros((2,2))\n", + "C[0,0]= cov_xx/variance_x\n", + "C[1,1]= cov_yy/variance_y\n", + "C[0,1]= cov_xy/np.sqrt(variance_y*variance_x)\n", + "C[1,0]= C[0,1]\n", + "print(C)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We see that the matrix elements along the diagonal are one as they\n", + "should be and that the matrix is symmetric. Furthermore, diagonalizing\n", + "this matrix we easily see that it is a positive definite matrix.\n", + "\n", + "The above procedure with **numpy** can be made more compact if we use **pandas**.\n", + "\n", + "\n", + "We whow here how we can set up the correlation matrix using **pandas**, as done in this simple code" + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "[[-1.0123897 -2.8039812 ]\n", + " [-0.28718135 -0.92440839]\n", + " [-0.74988099 -2.62660334]\n", + " [-0.10484789 -0.22717222]\n", + " [-0.62761155 -1.59313927]\n", + " [ 0.99624087 3.35932784]\n", + " [-0.26202974 -1.03435975]\n", + " [-0.36630178 -0.10312915]\n", + " [-0.4877836 -1.72320998]\n", + " [ 2.90178573 7.67667546]]\n", + " 0 1\n", + "0 -1.012390 -2.803981\n", + "1 -0.287181 -0.924408\n", + "2 -0.749881 -2.626603\n", + "3 -0.104848 -0.227172\n", + "4 -0.627612 -1.593139\n", + "5 0.996241 3.359328\n", + "6 -0.262030 -1.034360\n", + "7 -0.366302 -0.103129\n", + "8 -0.487784 -1.723210\n", + "9 2.901786 7.676675\n", + " 0 1\n", + "0 1.000000 0.989696\n", + "1 0.989696 1.000000\n" + ] + } + ], + "source": [ + "import numpy as np\n", + "import pandas as pd\n", + "n = 10\n", + "x = np.random.normal(size=n)\n", + "x = x - np.mean(x)\n", + "y = 4+3*x+np.random.normal(size=n)\n", + "y = y - np.mean(y)\n", + "X = (np.vstack((x, y))).T\n", + "print(X)\n", + "Xpd = pd.DataFrame(X)\n", + "print(Xpd)\n", + "correlation_matrix = Xpd.corr()\n", + "print(correlation_matrix)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We expand this model to the Franke function discussed above." + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + " 0 1 2 3 4 5 6 7 \\\n", + "0 0.0 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000 \n", + "1 0.0 0.075348 0.082602 0.073564 0.078039 0.082574 0.065469 0.068545 \n", + "2 0.0 0.082602 0.093536 0.081952 0.088312 0.094670 0.072999 0.077152 \n", + "3 0.0 0.073564 0.081952 0.077804 0.082702 0.087489 0.072629 0.075932 \n", + "4 0.0 0.078039 0.088312 0.082702 0.088620 0.094412 0.076979 0.080887 \n", + "5 0.0 0.082574 0.094670 0.087489 0.094412 0.101207 0.081135 0.085647 \n", + "6 0.0 0.065469 0.072999 0.072629 0.076979 0.081135 0.069875 0.072824 \n", + "7 0.0 0.068545 0.077152 0.075932 0.080887 0.085647 0.072824 0.076146 \n", + "8 0.0 0.071758 0.081453 0.079293 0.084869 0.090255 0.075775 0.079482 \n", + "9 0.0 0.075149 0.085972 0.082762 0.088986 0.095035 0.078769 0.082882 \n", + "10 0.0 0.057678 0.063987 0.065995 0.069640 0.073058 0.064814 0.067321 \n", + "11 0.0 0.059991 0.066970 0.068481 0.072514 0.076323 0.067071 0.069824 \n", + "12 0.0 0.062435 0.070113 0.071062 0.075503 0.079726 0.069384 0.072399 \n", + "13 0.0 0.065032 0.073451 0.073761 0.078639 0.083307 0.071774 0.075069 \n", + "14 0.0 0.067805 0.077022 0.076602 0.081951 0.087104 0.074261 0.077858 \n", + "\n", + " 8 9 10 11 12 13 14 \n", + "0 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000 0.000000 \n", + "1 0.071758 0.075149 0.057678 0.059991 0.062435 0.065032 0.067805 \n", + "2 0.081453 0.085972 0.063987 0.066970 0.070113 0.073451 0.077022 \n", + "3 0.079293 0.082762 0.065995 0.068481 0.071062 0.073761 0.076602 \n", + "4 0.084869 0.088986 0.069640 0.072514 0.075503 0.078639 0.081951 \n", + "5 0.090255 0.095035 0.073058 0.076323 0.079726 0.083307 0.087104 \n", + "6 0.075775 0.078769 0.064814 0.067071 0.069384 0.071774 0.074261 \n", + "7 0.079482 0.082882 0.067321 0.069824 0.072399 0.075069 0.077858 \n", + "8 0.083219 0.087047 0.069793 0.072554 0.075401 0.078366 0.081476 \n", + "9 0.087047 0.091332 0.072268 0.075299 0.078437 0.081717 0.085171 \n", + "10 0.069793 0.072268 0.061016 0.062978 0.064967 0.067002 0.069095 \n", + "11 0.072554 0.075299 0.062978 0.065108 0.067277 0.069504 0.071804 \n", + "12 0.075401 0.078437 0.064967 0.067277 0.069637 0.072069 0.074593 \n", + "13 0.078366 0.081717 0.067002 0.069504 0.072069 0.074724 0.077490 \n", + "14 0.081476 0.085171 0.069095 0.071804 0.074593 0.077490 0.080521 \n" + ] + } + ], + "source": [ + "# Common imports\n", + "import numpy as np\n", + "import pandas as pd\n", + "\n", + "\n", + "def FrankeFunction(x,y):\n", + "\tterm1 = 0.75*np.exp(-(0.25*(9*x-2)**2) - 0.25*((9*y-2)**2))\n", + "\tterm2 = 0.75*np.exp(-((9*x+1)**2)/49.0 - 0.1*(9*y+1))\n", + "\tterm3 = 0.5*np.exp(-(9*x-7)**2/4.0 - 0.25*((9*y-3)**2))\n", + "\tterm4 = -0.2*np.exp(-(9*x-4)**2 - (9*y-7)**2)\n", + "\treturn term1 + term2 + term3 + term4\n", + "\n", + "\n", + "def create_X(x, y, n ):\n", + "\tif len(x.shape) > 1:\n", + "\t\tx = np.ravel(x)\n", + "\t\ty = np.ravel(y)\n", + "\n", + "\tN = len(x)\n", + "\tl = int((n+1)*(n+2)/2)\t\t# Number of elements in beta\n", + "\tX = np.ones((N,l))\n", + "\n", + "\tfor i in range(1,n+1):\n", + "\t\tq = int((i)*(i+1)/2)\n", + "\t\tfor k in range(i+1):\n", + "\t\t\tX[:,q+k] = (x**(i-k))*(y**k)\n", + "\n", + "\treturn X\n", + "\n", + "\n", + "# Making meshgrid of datapoints and compute Franke's function\n", + "n = 4\n", + "N = 100\n", + "x = np.sort(np.random.uniform(0, 1, N))\n", + "y = np.sort(np.random.uniform(0, 1, N))\n", + "z = FrankeFunction(x, y)\n", + "X = create_X(x, y, n=n) \n", + "\n", + "Xpd = pd.DataFrame(X)\n", + "# subtract the mean values and set up the covariance matrix\n", + "Xpd = Xpd - Xpd.mean()\n", + "covariance_matrix = Xpd.cov()\n", + "print(covariance_matrix)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We note here that the covariance is zero for the first rows and\n", + "columns since all matrix elements in the design matrix were set to one\n", + "(we are fitting the function in terms of a polynomial of degree $n$).\n", + "\n", + "This means that the variance for these elements will be zero and will\n", + "cause problems when we set up the correlation matrix. We can simply\n", + "drop these elements and construct a correlation\n", + "matrix without these elements. \n", + "\n", + "\n", + "\n", + "\n", + "We can rewrite the covariance matrix in a more compact form in terms of the design/feature matrix $\\boldsymbol{X}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x}] = \\frac{1}{n}\\boldsymbol{X}^T\\boldsymbol{X}= \\mathbb{E}[\\boldsymbol{X}^T\\boldsymbol{X}].\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "To see this let us simply look at a design matrix $\\boldsymbol{X}\\in {\\mathbb{R}}^{2\\times 2}$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\\begin{bmatrix}\n", + "x_{00} & x_{01}\\\\\n", + "x_{10} & x_{11}\\\\\n", + "\\end{bmatrix}=\\begin{bmatrix}\n", + "\\boldsymbol{x}_{0} & \\boldsymbol{x}_{1}\\\\\n", + "\\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we then compute the expectation value" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}[\\boldsymbol{X}^T\\boldsymbol{X}] = \\frac{1}{n}\\boldsymbol{X}^T\\boldsymbol{X}=\\begin{bmatrix}\n", + "x_{00}^2+x_{01}^2 & x_{00}x_{10}+x_{01}x_{11}\\\\\n", + "x_{10}x_{00}+x_{11}x_{01} & x_{10}^2+x_{11}^2\\\\\n", + "\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which is just" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] = \\boldsymbol{C}[\\boldsymbol{x}]=\\begin{bmatrix} \\mathrm{var}[\\boldsymbol{x}_0] & \\mathrm{cov}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] \\\\\n", + " \\mathrm{cov}[\\boldsymbol{x}_1,\\boldsymbol{x}_0] & \\mathrm{var}[\\boldsymbol{x}_1] \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we wrote $$\\boldsymbol{C}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] = \\boldsymbol{C}[\\boldsymbol{x}]$$ to indicate that this the covariance of the vectors $\\boldsymbol{x}$ of the design/feature matrix $\\boldsymbol{X}$.\n", + "\n", + "It is easy to generalize this to a matrix $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$.\n", + "\n", + "\n", + "## Linking with SVD" + ] + } + ], + "metadata": { + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} \ No newline at end of file diff --git a/doc/LectureNotes/_build/jupyter_execute/chapter3.py b/doc/LectureNotes/_build/jupyter_execute/chapter3.py new file mode 100644 index 000000000..9469762fc --- /dev/null +++ b/doc/LectureNotes/_build/jupyter_execute/chapter3.py @@ -0,0 +1,790 @@ +# Ridge and Lasso Regression + +[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember10.mp4?vrtx=view-as-webpage) + + +## The singular value decomposition + +The examples we have looked at so far are cases where we normally can +invert the matrix $\boldsymbol{X}^T\boldsymbol{X}$. Using a polynomial expansion as we +did both for the masses and the fitting of the equation of state, +leads to row vectors of the design matrix which are essentially +orthogonal due to the polynomial character of our model. Obtaining the inverse of the design matrix is then often done via a so-called LU, QR or Cholesky decomposition. + + + +This may +however not the be case in general and a standard matrix inversion +algorithm based on say LU, QR or Cholesky decomposition may lead to singularities. We will see examples of this below. + +There is however a way to partially circumvent this problem and also gain some insights about the ordinary least squares approach, and later shrinkage methods like Ridge and Lasso regressions. + +This is given by the **Singular Value Decomposition** algorithm, perhaps +the most powerful linear algebra algorithm. Let us look at a +different example where we may have problems with the standard matrix +inversion algorithm. Thereafter we dive into the math of the SVD. + + + +One of the typical problems we encounter with linear regression, in particular +when the matrix $\boldsymbol{X}$ (our so-called design matrix) is high-dimensional, +are problems with near singular or singular matrices. The column vectors of $\boldsymbol{X}$ +may be linearly dependent, normally referred to as super-collinearity. +This means that the matrix may be rank deficient and it is basically impossible to +to model the data using linear regression. As an example, consider the matrix + +$$ +\begin{align*} +\mathbf{X} & = \left[ +\begin{array}{rrr} +1 & -1 & 2 +\\ +1 & 0 & 1 +\\ +1 & 2 & -1 +\\ +1 & 1 & 0 +\end{array} \right] +\end{align*} +$$ + +The columns of $\boldsymbol{X}$ are linearly dependent. We see this easily since the +the first column is the row-wise sum of the other two columns. The rank (more correct, +the column rank) of a matrix is the dimension of the space spanned by the +column vectors. Hence, the rank of $\mathbf{X}$ is equal to the number +of linearly independent columns. In this particular case the matrix has rank 2. + +Super-collinearity of an $(n \times p)$-dimensional design matrix $\mathbf{X}$ implies +that the inverse of the matrix $\boldsymbol{X}^T\boldsymbol{X}$ (the matrix we need to invert to solve the linear regression equations) is non-invertible. If we have a square matrix that does not have an inverse, we say this matrix singular. The example here demonstrates this + +$$ +\begin{align*} +\boldsymbol{X} & = \left[ +\begin{array}{rr} +1 & -1 +\\ +1 & -1 +\end{array} \right]. +\end{align*} +$$ + +We see easily that $\mbox{det}(\boldsymbol{X}) = x_{11} x_{22} - x_{12} x_{21} = 1 \times (-1) - 1 \times (-1) = 0$. Hence, $\mathbf{X}$ is singular and its inverse is undefined. +This is equivalent to saying that the matrix $\boldsymbol{X}$ has at least an eigenvalue which is zero. + + +If our design matrix $\boldsymbol{X}$ which enters the linear regression problem + + +
+ +$$ +\begin{equation} +\boldsymbol{\beta} = (\boldsymbol{X}^{T} \boldsymbol{X})^{-1} \boldsymbol{X}^{T} \boldsymbol{y}, +\label{_auto1} \tag{1} +\end{equation} +$$ + +has linearly dependent column vectors, we will not be able to compute the inverse +of $\boldsymbol{X}^T\boldsymbol{X}$ and we cannot find the parameters (estimators) $\beta_i$. +The estimators are only well-defined if $(\boldsymbol{X}^{T}\boldsymbol{X})^{-1}$ exits. +This is more likely to happen when the matrix $\boldsymbol{X}$ is high-dimensional. In this case it is likely to encounter a situation where +the regression parameters $\beta_i$ cannot be estimated. + +A cheap *ad hoc* approach is simply to add a small diagonal component to the matrix to invert, that is we change + +$$ +\boldsymbol{X}^{T} \boldsymbol{X} \rightarrow \boldsymbol{X}^{T} \boldsymbol{X}+\lambda \boldsymbol{I}, +$$ + +where $\boldsymbol{I}$ is the identity matrix. When we discuss **Ridge** regression this is actually what we end up evaluating. The parameter $\lambda$ is called a hyperparameter. More about this later. + + + + + +From standard linear algebra we know that a square matrix $\boldsymbol{X}$ can be diagonalized if and only it is +a so-called [normal matrix](https://en.wikipedia.org/wiki/Normal_matrix), that is if $\boldsymbol{X}\in {\mathbb{R}}^{n\times n}$ +we have $\boldsymbol{X}\boldsymbol{X}^T=\boldsymbol{X}^T\boldsymbol{X}$ or if $\boldsymbol{X}\in {\mathbb{C}}^{n\times n}$ we have $\boldsymbol{X}\boldsymbol{X}^{\dagger}=\boldsymbol{X}^{\dagger}\boldsymbol{X}$. +The matrix has then a set of eigenpairs + +$$ +(\lambda_1,\boldsymbol{u}_1),\dots, (\lambda_n,\boldsymbol{u}_n), +$$ + +and the eigenvalues are given by the diagonal matrix + +$$ +\boldsymbol{\Sigma}=\mathrm{Diag}(\lambda_1, \dots,\lambda_n). +$$ + +The matrix $\boldsymbol{X}$ can be written in terms of an orthogonal/unitary transformation $\boldsymbol{U}$ + +$$ +\boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T, +$$ + +with $\boldsymbol{U}\boldsymbol{U}^T=\boldsymbol{I}$ or $\boldsymbol{U}\boldsymbol{U}^{\dagger}=\boldsymbol{I}$. + +Not all square matrices are diagonalizable. A matrix like the one discussed above + +$$ +\boldsymbol{X} = \begin{bmatrix} +1& -1 \\ +1& -1\\ +\end{bmatrix} +$$ + +is not diagonalizable, it is a so-called [defective matrix](https://en.wikipedia.org/wiki/Defective_matrix). It is easy to see that the condition +$\boldsymbol{X}\boldsymbol{X}^T=\boldsymbol{X}^T\boldsymbol{X}$ is not fulfilled. + + + +## The SVD, a Fantastic Algorithm + + +However, and this is the strength of the SVD algorithm, any general +matrix $\boldsymbol{X}$ can be decomposed in terms of a diagonal matrix and +two orthogonal/unitary matrices. The [Singular Value Decompostion +(SVD) theorem](https://en.wikipedia.org/wiki/Singular_value_decomposition) +states that a general $m\times n$ matrix $\boldsymbol{X}$ can be written in +terms of a diagonal matrix $\boldsymbol{\Sigma}$ of dimensionality $m\times n$ +and two orthognal matrices $\boldsymbol{U}$ and $\boldsymbol{V}$, where the first has +dimensionality $m \times m$ and the last dimensionality $n\times n$. +We have then + +$$ +\boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T +$$ + +As an example, the above defective matrix can be decomposed as + +$$ +\boldsymbol{X} = \frac{1}{\sqrt{2}}\begin{bmatrix} 1& 1 \\ 1& -1\\ \end{bmatrix} \begin{bmatrix} 2& 0 \\ 0& 0\\ \end{bmatrix} \frac{1}{\sqrt{2}}\begin{bmatrix} 1& -1 \\ 1& 1\\ \end{bmatrix}=\boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T, +$$ + +with eigenvalues $\sigma_1=2$ and $\sigma_2=0$. +The SVD exits always! + +The SVD +decomposition (singular values) gives eigenvalues +$\sigma_i\geq\sigma_{i+1}$ for all $i$ and for dimensions larger than $i=p$, the +eigenvalues (singular values) are zero. + +In the general case, where our design matrix $\boldsymbol{X}$ has dimension +$n\times p$, the matrix is thus decomposed into an $n\times n$ +orthogonal matrix $\boldsymbol{U}$, a $p\times p$ orthogonal matrix $\boldsymbol{V}$ +and a diagonal matrix $\boldsymbol{\Sigma}$ with $r=\mathrm{min}(n,p)$ +singular values $\sigma_i\geq 0$ on the main diagonal and zeros filling +the rest of the matrix. There are at most $p$ singular values +assuming that $n > p$. In our regression examples for the nuclear +masses and the equation of state this is indeed the case, while for +the Ising model we have $p > n$. These are often cases that lead to +near singular or singular matrices. + +The columns of $\boldsymbol{U}$ are called the left singular vectors while the columns of $\boldsymbol{V}$ are the right singular vectors. + +## Economy-size SVD + +If we assume that $n > p$, then our matrix $\boldsymbol{U}$ has dimension $n +\times n$. The last $n-p$ columns of $\boldsymbol{U}$ become however +irrelevant in our calculations since they are multiplied with the +zeros in $\boldsymbol{\Sigma}$. + +The economy-size decomposition removes extra rows or columns of zeros +from the diagonal matrix of singular values, $\boldsymbol{\Sigma}$, along with the columns +in either $\boldsymbol{U}$ or $\boldsymbol{V}$ that multiply those zeros in the expression. +Removing these zeros and columns can improve execution time +and reduce storage requirements without compromising the accuracy of +the decomposition. + +If $n > p$, we keep only the first $p$ columns of $\boldsymbol{U}$ and $\boldsymbol{\Sigma}$ has dimension $p\times p$. +If $p > n$, then only the first $n$ columns of $\boldsymbol{V}$ are computed and $\boldsymbol{\Sigma}$ has dimension $n\times n$. +The $n=p$ case is obvious, we retain the full SVD. +In general the economy-size SVD leads to less FLOPS and still conserving the desired accuracy. + +import numpy as np +# SVD inversion +def SVDinv(A): + ''' Takes as input a numpy matrix A and returns inv(A) based on singular value decomposition (SVD). + SVD is numerically more stable than the inversion algorithms provided by + numpy and scipy.linalg at the cost of being slower. + ''' + U, s, VT = np.linalg.svd(A) +# print('test U') +# print( (np.transpose(U) @ U - U @np.transpose(U))) +# print('test VT') +# print( (np.transpose(VT) @ VT - VT @np.transpose(VT))) + print(U) + print(s) + print(VT) + + D = np.zeros((len(U),len(VT))) + for i in range(0,len(VT)): + D[i,i]=s[i] + UT = np.transpose(U); V = np.transpose(VT); invD = np.linalg.inv(D) + return np.matmul(V,np.matmul(invD,UT)) + + +X = np.array([ [1.0, -1.0, 2.0], [1.0, 0.0, 1.0], [1.0, 2.0, -1.0], [1.0, 1.0, 0.0] ]) +print(X) +A = np.transpose(X) @ X +print(A) +# Brute force inversion of super-collinear matrix +#B = np.linalg.inv(A) +#print(B) +C = SVDinv(A) +print(C) + +The matrix $\boldsymbol{X}$ has columns that are linearly dependent. The first +column is the row-wise sum of the other two columns. The rank of a +matrix (the column rank) is the dimension of space spanned by the +column vectors. The rank of the matrix is the number of linearly +independent columns, in this case just $2$. We see this from the +singular values when running the above code. Running the standard +inversion algorithm for matrix inversion with $\boldsymbol{X}^T\boldsymbol{X}$ results +in the program terminating due to a singular matrix. + + + + +There are several interesting mathematical properties which will be +relevant when we are going to discuss the differences between say +ordinary least squares (OLS) and **Ridge** regression. + +We have from OLS that the parameters of the linear approximation are given by + +$$ +\boldsymbol{\tilde{y}} = \boldsymbol{X}\boldsymbol{\beta} = \boldsymbol{X}\left(\boldsymbol{X}^T\boldsymbol{X}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}. +$$ + +The matrix to invert can be rewritten in terms of our SVD decomposition as + +$$ +\boldsymbol{X}^T\boldsymbol{X} = \boldsymbol{V}\boldsymbol{\Sigma}^T\boldsymbol{U}^T\boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T. +$$ + +Using the orthogonality properties of $\boldsymbol{U}$ we have + +$$ +\boldsymbol{X}^T\boldsymbol{X} = \boldsymbol{V}\boldsymbol{\Sigma}^T\boldsymbol{\Sigma}\boldsymbol{V}^T = \boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T, +$$ + +with $\boldsymbol{D}$ being a diagonal matrix with values along the diagonal given by the singular values squared. + +This means that + +$$ +(\boldsymbol{X}^T\boldsymbol{X})\boldsymbol{V} = \boldsymbol{V}\boldsymbol{D}, +$$ + +that is the eigenvectors of $(\boldsymbol{X}^T\boldsymbol{X})$ are given by the columns of the right singular matrix of $\boldsymbol{X}$ and the eigenvalues are the squared singular values. It is easy to show (show this) that + +$$ +(\boldsymbol{X}\boldsymbol{X}^T)\boldsymbol{U} = \boldsymbol{U}\boldsymbol{D}, +$$ + +that is, the eigenvectors of $(\boldsymbol{X}\boldsymbol{X})^T$ are the columns of the left singular matrix and the eigenvalues are the same. + +Going back to our OLS equation we have + +$$ +\boldsymbol{X}\boldsymbol{\beta} = \boldsymbol{X}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T \right)^{-1}\boldsymbol{X}^T\boldsymbol{y}=\boldsymbol{U\Sigma V^T}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T \right)^{-1}(\boldsymbol{U\Sigma V^T})^T\boldsymbol{y}=\boldsymbol{U}\boldsymbol{U}^T\boldsymbol{y}. +$$ + +We will come back to this expression when we discuss Ridge regression. + + +$$ \tilde{y}^{OLS}=\boldsymbol{X}\hat{\beta}^{OLS}=\sum_{j=1}^p \boldsymbol{u}_j\boldsymbol{u}_j^T\boldsymbol{y}$$ and for Ridge we have  + +$$ \tilde{y}^{Ridge}=\boldsymbol{X}\hat{\beta}^{Ridge}=\sum_{j=1}^p \boldsymbol{u}_j\frac{\sigma_j^2}{\sigma_j^2+\lambda}\boldsymbol{u}_j^T\boldsymbol{y}$$ .  + +It is indeed the economy-sized SVD, note the summation runs up tp $$p$$ only and not $$n$$.  + +Here we have that $$\boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma}\boldsymbol{V}^T$$, with $$\Sigma$$ being an $$ n\times p$$ matrix and $$\boldsymbol{V}$$ being a $$ p\times p$$ matrix. We also have assumed here that $$ n > p$$.  + + + + + + + + +## Ridge and LASSO Regression + +[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember11.mp4?vrtx=view-as-webpage) + +Let us remind ourselves about the expression for the standard Mean Squared Error (MSE) which we used to define our cost function and the equations for the ordinary least squares (OLS) method, that is +our optimization problem is + +$$ +{\displaystyle \min_{\boldsymbol{\beta}\in {\mathbb{R}}^{p}}}\frac{1}{n}\left\{\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)^T\left(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\right)\right\}. +$$ + +or we can state it as + +$$ +{\displaystyle \min_{\boldsymbol{\beta}\in +{\mathbb{R}}^{p}}}\frac{1}{n}\sum_{i=0}^{n-1}\left(y_i-\tilde{y}_i\right)^2=\frac{1}{n}\vert\vert \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\vert\vert_2^2, +$$ + +where we have used the definition of a norm-2 vector, that is + +$$ +\vert\vert \boldsymbol{x}\vert\vert_2 = \sqrt{\sum_i x_i^2}. +$$ + +By minimizing the above equation with respect to the parameters +$\boldsymbol{\beta}$ we could then obtain an analytical expression for the +parameters $\boldsymbol{\beta}$. We can add a regularization parameter $\lambda$ by +defining a new cost function to be optimized, that is + +$$ +{\displaystyle \min_{\boldsymbol{\beta}\in +{\mathbb{R}}^{p}}}\frac{1}{n}\vert\vert \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\vert\vert_2^2+\lambda\vert\vert \boldsymbol{\beta}\vert\vert_2^2 +$$ + +which leads to the Ridge regression minimization problem where we +require that $\vert\vert \boldsymbol{\beta}\vert\vert_2^2\le t$, where $t$ is +a finite number larger than zero. By defining + +$$ +C(\boldsymbol{X},\boldsymbol{\beta})=\frac{1}{n}\vert\vert \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\vert\vert_2^2+\lambda\vert\vert \boldsymbol{\beta}\vert\vert_1, +$$ + +we have a new optimization equation + +$$ +{\displaystyle \min_{\boldsymbol{\beta}\in +{\mathbb{R}}^{p}}}\frac{1}{n}\vert\vert \boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta}\vert\vert_2^2+\lambda\vert\vert \boldsymbol{\beta}\vert\vert_1 +$$ + +which leads to Lasso regression. Lasso stands for least absolute shrinkage and selection operator. + +Here we have defined the norm-1 as + +$$ +\vert\vert \boldsymbol{x}\vert\vert_1 = \sum_i \vert x_i\vert. +$$ + +Using the matrix-vector expression for Ridge regression, + +$$ +C(\boldsymbol{X},\boldsymbol{\beta})=\frac{1}{n}\left\{(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta})^T(\boldsymbol{y}-\boldsymbol{X}\boldsymbol{\beta})\right\}+\lambda\boldsymbol{\beta}^T\boldsymbol{\beta}, +$$ + +by taking the derivatives with respect to $\boldsymbol{\beta}$ we obtain then +a slightly modified matrix inversion problem which for finite values +of $\lambda$ does not suffer from singularity problems. We obtain + +$$ +\boldsymbol{\beta}^{\mathrm{Ridge}} = \left(\boldsymbol{X}^T\boldsymbol{X}+\lambda\boldsymbol{I}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}, +$$ + +with $\boldsymbol{I}$ being a $p\times p$ identity matrix with the constraint that + +$$ +\sum_{i=0}^{p-1} \beta_i^2 \leq t, +$$ + +with $t$ a finite positive number. + +We see that Ridge regression is nothing but the standard +OLS with a modified diagonal term added to $\boldsymbol{X}^T\boldsymbol{X}$. The +consequences, in particular for our discussion of the bias-variance tradeoff +are rather interesting. + +Furthermore, if we use the result above in terms of the SVD decomposition (our analysis was done for the OLS method), we had + +$$ +(\boldsymbol{X}\boldsymbol{X}^T)\boldsymbol{U} = \boldsymbol{U}\boldsymbol{D}. +$$ + +We can analyse the OLS solutions in terms of the eigenvectors (the columns) of the right singular value matrix $\boldsymbol{U}$ as + +$$ +\boldsymbol{X}\boldsymbol{\beta} = \boldsymbol{X}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T \right)^{-1}\boldsymbol{X}^T\boldsymbol{y}=\boldsymbol{U\Sigma V^T}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T \right)^{-1}(\boldsymbol{U\Sigma V^T})^T\boldsymbol{y}=\boldsymbol{U}\boldsymbol{U}^T\boldsymbol{y} +$$ + +For Ridge regression this becomes + +$$ +\boldsymbol{X}\boldsymbol{\beta}^{\mathrm{Ridge}} = \boldsymbol{U\Sigma V^T}\left(\boldsymbol{V}\boldsymbol{D}\boldsymbol{V}^T+\lambda\boldsymbol{I} \right)^{-1}(\boldsymbol{U\Sigma V^T})^T\boldsymbol{y}=\sum_{j=0}^{p-1}\boldsymbol{u}_j\boldsymbol{u}_j^T\frac{\sigma_j^2}{\sigma_j^2+\lambda}\boldsymbol{y}, +$$ + +with the vectors $\boldsymbol{u}_j$ being the columns of $\boldsymbol{U}$. + + +Since $\lambda \geq 0$, it means that compared to OLS, we have + +$$ +\frac{\sigma_j^2}{\sigma_j^2+\lambda} \leq 1. +$$ + +Ridge regression finds the coordinates of $\boldsymbol{y}$ with respect to the +orthonormal basis $\boldsymbol{U}$, it then shrinks the coordinates by +$\frac{\sigma_j^2}{\sigma_j^2+\lambda}$. Recall that the SVD has +eigenvalues ordered in a descending way, that is $\sigma_i \geq +\sigma_{i+1}$. + +For small eigenvalues $\sigma_i$ it means that their contributions become less important, a fact which can be used to reduce the number of degrees of freedom. +Actually, calculating the variance of $\boldsymbol{X}\boldsymbol{v}_j$ shows that this quantity is equal to $\sigma_j^2/n$. +With a parameter $\lambda$ we can thus shrink the role of specific parameters. + + + +For the sake of simplicity, let us assume that the design matrix is orthonormal, that is + +$$ +\boldsymbol{X}^T\boldsymbol{X}=(\boldsymbol{X}^T\boldsymbol{X})^{-1} =\boldsymbol{I}. +$$ + +In this case the standard OLS results in + +$$ +\boldsymbol{\beta}^{\mathrm{OLS}} = \boldsymbol{X}^T\boldsymbol{y}=\sum_{i=0}^{p-1}\boldsymbol{u}_j\boldsymbol{u}_j^T\boldsymbol{y}, +$$ + +and + +$$ +\boldsymbol{\beta}^{\mathrm{Ridge}} = \left(\boldsymbol{I}+\lambda\boldsymbol{I}\right)^{-1}\boldsymbol{X}^T\boldsymbol{y}=\left(1+\lambda\right)^{-1}\boldsymbol{\beta}^{\mathrm{OLS}}, +$$ + +that is the Ridge estimator scales the OLS estimator by the inverse of a factor $1+\lambda$, and +the Ridge estimator converges to zero when the hyperparameter goes to +infinity. + +We will come back to more interpreations after we have gone through some of the statistical analysis part. + +For more discussions of Ridge and Lasso regression, [Wessel van Wieringen's](https://arxiv.org/abs/1509.09169) article is highly recommended. +Similarly, [Mehta et al's article](https://arxiv.org/abs/1803.08823) is also recommended. + + + +## A better understanding of regularization + +The parameter $\lambda$ that we have introduced in the Ridge (and +Lasso as well) regression is often called a regularization parameter +or shrinkage parameter. It is common to call it a hyperparameter. What does it mean mathemtically? + +Here we will first look at how to analyze the difference between the +standard OLS equations and the Ridge expressions in terms of a linear +algebra analysis using the SVD algorithm. Thereafter, we will link +(see the material on the bias-variance tradeoff below) these +observation to the statisical analysis of the results. In particular +we consider how the variance of the parameters $\boldsymbol{\beta}$ is +affected by changing the parameter $\lambda$. + + +We have our design matrix + $\boldsymbol{X}\in {\mathbb{R}}^{n\times p}$. With the SVD we decompose it as + +$$ +\boldsymbol{X} = \boldsymbol{U\Sigma V^T}, +$$ + +with $\boldsymbol{U}\in {\mathbb{R}}^{n\times n}$, $\boldsymbol{\Sigma}\in {\mathbb{R}}^{n\times p}$ +and $\boldsymbol{V}\in {\mathbb{R}}^{p\times p}$. + +The matrices $\boldsymbol{U}$ and $\boldsymbol{V}$ are unitary/orthonormal matrices, that is in case the matrices are real we have $\boldsymbol{U}^T\boldsymbol{U}=\boldsymbol{U}\boldsymbol{U}^T=\boldsymbol{I}$ and $\boldsymbol{V}^T\boldsymbol{V}=\boldsymbol{V}\boldsymbol{V}^T=\boldsymbol{I}$. + + + +## Introducing the Covariance and Correlation functions + +Before we discuss the link between for example Ridge regression and the singular value decomposition, we need to remind ourselves about +the definition of the covariance and the correlation function. These are quantities + +Suppose we have defined two vectors +$\hat{x}$ and $\hat{y}$ with $n$ elements each. The covariance matrix $\boldsymbol{C}$ is defined as + +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} \mathrm{cov}[\boldsymbol{x},\boldsymbol{x}] & \mathrm{cov}[\boldsymbol{x},\boldsymbol{y}] \\ + \mathrm{cov}[\boldsymbol{y},\boldsymbol{x}] & \mathrm{cov}[\boldsymbol{y},\boldsymbol{y}] \\ + \end{bmatrix}, +$$ + +where for example + +$$ +\mathrm{cov}[\boldsymbol{x},\boldsymbol{y}] =\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})(y_i- \overline{y}). +$$ + +With this definition and recalling that the variance is defined as + +$$ +\mathrm{var}[\boldsymbol{x}]=\frac{1}{n} \sum_{i=0}^{n-1}(x_i- \overline{x})^2, +$$ + +we can rewrite the covariance matrix as + +$$ +\boldsymbol{C}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} \mathrm{var}[\boldsymbol{x}] & \mathrm{cov}[\boldsymbol{x},\boldsymbol{y}] \\ + \mathrm{cov}[\boldsymbol{x},\boldsymbol{y}] & \mathrm{var}[\boldsymbol{y}] \\ + \end{bmatrix}. +$$ + +The covariance takes values between zero and infinity and may thus +lead to problems with loss of numerical precision for particularly +large values. It is common to scale the covariance matrix by +introducing instead the correlation matrix defined via the so-called +correlation function + +$$ +\mathrm{corr}[\boldsymbol{x},\boldsymbol{y}]=\frac{\mathrm{cov}[\boldsymbol{x},\boldsymbol{y}]}{\sqrt{\mathrm{var}[\boldsymbol{x}] \mathrm{var}[\boldsymbol{y}]}}. +$$ + +The correlation function is then given by values $\mathrm{corr}[\boldsymbol{x},\boldsymbol{y}] +\in [-1,1]$. This avoids eventual problems with too large values. We +can then define the correlation matrix for the two vectors $\boldsymbol{x}$ +and $\boldsymbol{y}$ as + +$$ +\boldsymbol{K}[\boldsymbol{x},\boldsymbol{y}] = \begin{bmatrix} 1 & \mathrm{corr}[\boldsymbol{x},\boldsymbol{y}] \\ + \mathrm{corr}[\boldsymbol{y},\boldsymbol{x}] & 1 \\ + \end{bmatrix}, +$$ + +In the above example this is the function we constructed using **pandas**. + + + +In our derivation of the various regression algorithms like **Ordinary Least Squares** or **Ridge regression** +we defined the design/feature matrix $\boldsymbol{X}$ as + +$$ +\boldsymbol{X}=\begin{bmatrix} +x_{0,0} & x_{0,1} & x_{0,2}& \dots & \dots x_{0,p-1}\\ +x_{1,0} & x_{1,1} & x_{1,2}& \dots & \dots x_{1,p-1}\\ +x_{2,0} & x_{2,1} & x_{2,2}& \dots & \dots x_{2,p-1}\\ +\dots & \dots & \dots & \dots \dots & \dots \\ +x_{n-2,0} & x_{n-2,1} & x_{n-2,2}& \dots & \dots x_{n-2,p-1}\\ +x_{n-1,0} & x_{n-1,1} & x_{n-1,2}& \dots & \dots x_{n-1,p-1}\\ +\end{bmatrix}, +$$ + +with $\boldsymbol{X}\in {\mathbb{R}}^{n\times p}$, with the predictors/features $p$ refering to the column numbers and the +entries $n$ being the row elements. +We can rewrite the design/feature matrix in terms of its column vectors as + +$$ +\boldsymbol{X}=\begin{bmatrix} \boldsymbol{x}_0 & \boldsymbol{x}_1 & \boldsymbol{x}_2 & \dots & \dots & \boldsymbol{x}_{p-1}\end{bmatrix}, +$$ + +with a given vector + +$$ +\boldsymbol{x}_i^T = \begin{bmatrix}x_{0,i} & x_{1,i} & x_{2,i}& \dots & \dots x_{n-1,i}\end{bmatrix}. +$$ + +With these definitions, we can now rewrite our $2\times 2$ +correaltion/covariance matrix in terms of a moe general design/feature +matrix $\boldsymbol{X}\in {\mathbb{R}}^{n\times p}$. This leads to a $p\times p$ +covariance matrix for the vectors $\boldsymbol{x}_i$ with $i=0,1,\dots,p-1$ + +$$ +\boldsymbol{C}[\boldsymbol{x}] = \begin{bmatrix} +\mathrm{var}[\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_1] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_2] & \dots & \dots & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_{p-1}]\\ +\mathrm{cov}[\boldsymbol{x}_1,\boldsymbol{x}_0] & \mathrm{var}[\boldsymbol{x}_1] & \mathrm{cov}[\boldsymbol{x}_1,\boldsymbol{x}_2] & \dots & \dots & \mathrm{cov}[\boldsymbol{x}_1,\boldsymbol{x}_{p-1}]\\ +\mathrm{cov}[\boldsymbol{x}_2,\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_2,\boldsymbol{x}_1] & \mathrm{var}[\boldsymbol{x}_2] & \dots & \dots & \mathrm{cov}[\boldsymbol{x}_2,\boldsymbol{x}_{p-1}]\\ +\dots & \dots & \dots & \dots & \dots & \dots \\ +\dots & \dots & \dots & \dots & \dots & \dots \\ +\mathrm{cov}[\boldsymbol{x}_{p-1},\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_{p-1},\boldsymbol{x}_1] & \mathrm{cov}[\boldsymbol{x}_{p-1},\boldsymbol{x}_{2}] & \dots & \dots & \mathrm{var}[\boldsymbol{x}_{p-1}]\\ +\end{bmatrix}, +$$ + +and the correlation matrix + +$$ +\boldsymbol{K}[\boldsymbol{x}] = \begin{bmatrix} +1 & \mathrm{corr}[\boldsymbol{x}_0,\boldsymbol{x}_1] & \mathrm{corr}[\boldsymbol{x}_0,\boldsymbol{x}_2] & \dots & \dots & \mathrm{corr}[\boldsymbol{x}_0,\boldsymbol{x}_{p-1}]\\ +\mathrm{corr}[\boldsymbol{x}_1,\boldsymbol{x}_0] & 1 & \mathrm{corr}[\boldsymbol{x}_1,\boldsymbol{x}_2] & \dots & \dots & \mathrm{corr}[\boldsymbol{x}_1,\boldsymbol{x}_{p-1}]\\ +\mathrm{corr}[\boldsymbol{x}_2,\boldsymbol{x}_0] & \mathrm{corr}[\boldsymbol{x}_2,\boldsymbol{x}_1] & 1 & \dots & \dots & \mathrm{corr}[\boldsymbol{x}_2,\boldsymbol{x}_{p-1}]\\ +\dots & \dots & \dots & \dots & \dots & \dots \\ +\dots & \dots & \dots & \dots & \dots & \dots \\ +\mathrm{corr}[\boldsymbol{x}_{p-1},\boldsymbol{x}_0] & \mathrm{corr}[\boldsymbol{x}_{p-1},\boldsymbol{x}_1] & \mathrm{corr}[\boldsymbol{x}_{p-1},\boldsymbol{x}_{2}] & \dots & \dots & 1\\ +\end{bmatrix}, +$$ + +The Numpy function **np.cov** calculates the covariance elements using +the factor $1/(n-1)$ instead of $1/n$ since it assumes we do not have +the exact mean values. The following simple function uses the +**np.vstack** function which takes each vector of dimension $1\times n$ +and produces a $2\times n$ matrix $\boldsymbol{W}$ + +$$ +\boldsymbol{W} = \begin{bmatrix} x_0 & y_0 \\ + x_1 & y_1 \\ + x_2 & y_2\\ + \dots & \dots \\ + x_{n-2} & y_{n-2}\\ + x_{n-1} & y_{n-1} & + \end{bmatrix}, +$$ + +which in turn is converted into into the $2\times 2$ covariance matrix +$\boldsymbol{C}$ via the Numpy function **np.cov()**. We note that we can also calculate +the mean value of each set of samples $\boldsymbol{x}$ etc using the Numpy +function **np.mean(x)**. We can also extract the eigenvalues of the +covariance matrix through the **np.linalg.eig()** function. + +# Importing various packages +import numpy as np +n = 100 +x = np.random.normal(size=n) +print(np.mean(x)) +y = 4+3*x+np.random.normal(size=n) +print(np.mean(y)) +W = np.vstack((x, y)) +C = np.cov(W) +print(C) + +The previous example can be converted into the correlation matrix by +simply scaling the matrix elements with the variances. We should also +subtract the mean values for each column. This leads to the following +code which sets up the correlations matrix for the previous example in +a more brute force way. Here we scale the mean values for each column of the design matrix, calculate the relevant mean values and variances and then finally set up the $2\times 2$ correlation matrix (since we have only two vectors). + +import numpy as np +n = 100 +# define two vectors +x = np.random.random(size=n) +y = 4+3*x+np.random.normal(size=n) +#scaling the x and y vectors +x = x - np.mean(x) +y = y - np.mean(y) +variance_x = np.sum(x@x)/n +variance_y = np.sum(y@y)/n +print(variance_x) +print(variance_y) +cov_xy = np.sum(x@y)/n +cov_xx = np.sum(x@x)/n +cov_yy = np.sum(y@y)/n +C = np.zeros((2,2)) +C[0,0]= cov_xx/variance_x +C[1,1]= cov_yy/variance_y +C[0,1]= cov_xy/np.sqrt(variance_y*variance_x) +C[1,0]= C[0,1] +print(C) + +We see that the matrix elements along the diagonal are one as they +should be and that the matrix is symmetric. Furthermore, diagonalizing +this matrix we easily see that it is a positive definite matrix. + +The above procedure with **numpy** can be made more compact if we use **pandas**. + + +We whow here how we can set up the correlation matrix using **pandas**, as done in this simple code + +import numpy as np +import pandas as pd +n = 10 +x = np.random.normal(size=n) +x = x - np.mean(x) +y = 4+3*x+np.random.normal(size=n) +y = y - np.mean(y) +X = (np.vstack((x, y))).T +print(X) +Xpd = pd.DataFrame(X) +print(Xpd) +correlation_matrix = Xpd.corr() +print(correlation_matrix) + +We expand this model to the Franke function discussed above. + +# Common imports +import numpy as np +import pandas as pd + + +def FrankeFunction(x,y): + term1 = 0.75*np.exp(-(0.25*(9*x-2)**2) - 0.25*((9*y-2)**2)) + term2 = 0.75*np.exp(-((9*x+1)**2)/49.0 - 0.1*(9*y+1)) + term3 = 0.5*np.exp(-(9*x-7)**2/4.0 - 0.25*((9*y-3)**2)) + term4 = -0.2*np.exp(-(9*x-4)**2 - (9*y-7)**2) + return term1 + term2 + term3 + term4 + + +def create_X(x, y, n ): + if len(x.shape) > 1: + x = np.ravel(x) + y = np.ravel(y) + + N = len(x) + l = int((n+1)*(n+2)/2) # Number of elements in beta + X = np.ones((N,l)) + + for i in range(1,n+1): + q = int((i)*(i+1)/2) + for k in range(i+1): + X[:,q+k] = (x**(i-k))*(y**k) + + return X + + +# Making meshgrid of datapoints and compute Franke's function +n = 4 +N = 100 +x = np.sort(np.random.uniform(0, 1, N)) +y = np.sort(np.random.uniform(0, 1, N)) +z = FrankeFunction(x, y) +X = create_X(x, y, n=n) + +Xpd = pd.DataFrame(X) +# subtract the mean values and set up the covariance matrix +Xpd = Xpd - Xpd.mean() +covariance_matrix = Xpd.cov() +print(covariance_matrix) + +We note here that the covariance is zero for the first rows and +columns since all matrix elements in the design matrix were set to one +(we are fitting the function in terms of a polynomial of degree $n$). + +This means that the variance for these elements will be zero and will +cause problems when we set up the correlation matrix. We can simply +drop these elements and construct a correlation +matrix without these elements. + + + + +We can rewrite the covariance matrix in a more compact form in terms of the design/feature matrix $\boldsymbol{X}$ as + +$$ +\boldsymbol{C}[\boldsymbol{x}] = \frac{1}{n}\boldsymbol{X}^T\boldsymbol{X}= \mathbb{E}[\boldsymbol{X}^T\boldsymbol{X}]. +$$ + +To see this let us simply look at a design matrix $\boldsymbol{X}\in {\mathbb{R}}^{2\times 2}$ + +$$ +\boldsymbol{X}=\begin{bmatrix} +x_{00} & x_{01}\\ +x_{10} & x_{11}\\ +\end{bmatrix}=\begin{bmatrix} +\boldsymbol{x}_{0} & \boldsymbol{x}_{1}\\ +\end{bmatrix}. +$$ + +If we then compute the expectation value + +$$ +\mathbb{E}[\boldsymbol{X}^T\boldsymbol{X}] = \frac{1}{n}\boldsymbol{X}^T\boldsymbol{X}=\begin{bmatrix} +x_{00}^2+x_{01}^2 & x_{00}x_{10}+x_{01}x_{11}\\ +x_{10}x_{00}+x_{11}x_{01} & x_{10}^2+x_{11}^2\\ +\end{bmatrix}, +$$ + +which is just + +$$ +\boldsymbol{C}[\boldsymbol{x}_0,\boldsymbol{x}_1] = \boldsymbol{C}[\boldsymbol{x}]=\begin{bmatrix} \mathrm{var}[\boldsymbol{x}_0] & \mathrm{cov}[\boldsymbol{x}_0,\boldsymbol{x}_1] \\ + \mathrm{cov}[\boldsymbol{x}_1,\boldsymbol{x}_0] & \mathrm{var}[\boldsymbol{x}_1] \\ + \end{bmatrix}, +$$ + +where we wrote $$\boldsymbol{C}[\boldsymbol{x}_0,\boldsymbol{x}_1] = \boldsymbol{C}[\boldsymbol{x}]$$ to indicate that this the covariance of the vectors $\boldsymbol{x}$ of the design/feature matrix $\boldsymbol{X}$. + +It is easy to generalize this to a matrix $\boldsymbol{X}\in {\mathbb{R}}^{n\times p}$. + + +## Linking with SVD \ No newline at end of file diff --git a/doc/LectureNotes/_build/jupyter_execute/chapter4.ipynb b/doc/LectureNotes/_build/jupyter_execute/chapter4.ipynb new file mode 100644 index 000000000..83057b090 --- /dev/null +++ b/doc/LectureNotes/_build/jupyter_execute/chapter4.ipynb @@ -0,0 +1,2900 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Logistic Regression\n", + "\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK3155/h20/forelesningsvideoer/LectureSeptember18.mp4?vrtx=view-as-webpage)\n", + "\n", + "\n", + "## Logistic Regression\n", + "\n", + "In linear regression our main interest was centered on learning the\n", + "coefficients of a functional fit (say a polynomial) in order to be\n", + "able to predict the response of a continuous variable on some unseen\n", + "data. The fit to the continuous variable $y_i$ is based on some\n", + "independent variables $\\hat{x}_i$. Linear regression resulted in\n", + "analytical expressions for standard ordinary Least Squares or Ridge\n", + "regression (in terms of matrices to invert) for several quantities,\n", + "ranging from the variance and thereby the confidence intervals of the\n", + "parameters $\\hat{\\beta}$ to the mean squared error. If we can invert\n", + "the product of the design matrices, linear regression gives then a\n", + "simple recipe for fitting our data.\n", + "\n", + "\n", + "Classification problems, however, are concerned with outcomes taking\n", + "the form of discrete variables (i.e. categories). We may for example,\n", + "on the basis of DNA sequencing for a number of patients, like to find\n", + "out which mutations are important for a certain disease; or based on\n", + "scans of various patients' brains, figure out if there is a tumor or\n", + "not; or given a specific physical system, we'd like to identify its\n", + "state, say whether it is an ordered or disordered system (typical\n", + "situation in solid state physics); or classify the status of a\n", + "patient, whether she/he has a stroke or not and many other similar\n", + "situations.\n", + "\n", + "The most common situation we encounter when we apply logistic\n", + "regression is that of two possible outcomes, normally denoted as a\n", + "binary outcome, true or false, positive or negative, success or\n", + "failure etc.\n", + "\n", + "\n", + "Logistic regression will also serve as our stepping stone towards\n", + "neural network algorithms and supervised deep learning. For logistic\n", + "learning, the minimization of the cost function leads to a non-linear\n", + "equation in the parameters $\\hat{\\beta}$. The optimization of the\n", + "problem calls therefore for minimization algorithms. This forms the\n", + "bottle neck of all machine learning algorithms, namely how to find\n", + "reliable minima of a multi-variable function. This leads us to the\n", + "family of gradient descent methods. The latter are the working horses\n", + "of basically all modern machine learning algorithms.\n", + "\n", + "We note also that many of the topics discussed here on logistic \n", + "regression are also commonly used in modern supervised Deep Learning\n", + "models, as we will see later.\n", + "\n", + "\n", + "\n", + "## Basics\n", + "\n", + "We consider the case where the dependent variables, also called the\n", + "responses or the outcomes, $y_i$ are discrete and only take values\n", + "from $k=0,\\dots,K-1$ (i.e. $K$ classes).\n", + "\n", + "The goal is to predict the\n", + "output classes from the design matrix $\\hat{X}\\in\\mathbb{R}^{n\\times p}$\n", + "made of $n$ samples, each of which carries $p$ features or predictors. The\n", + "primary goal is to identify the classes to which new unseen samples\n", + "belong.\n", + "\n", + "Let us specialize to the case of two classes only, with outputs\n", + "$y_i=0$ and $y_i=1$. Our outcomes could represent the status of a\n", + "credit card user that could default or not on her/his credit card\n", + "debt. That is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y_i = \\begin{bmatrix} 0 & \\mathrm{no}\\\\ 1 & \\mathrm{yes} \\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Before moving to the logistic model, let us try to use our linear\n", + "regression model to classify these two outcomes. We could for example\n", + "fit a linear model to the default case if $y_i > 0.5$ and the no\n", + "default case $y_i \\leq 0.5$.\n", + "\n", + "We would then have our \n", + "weighted linear combination, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "
\n", + "\n", + "$$\n", + "\\begin{equation}\n", + "\\hat{y} = \\hat{X}^T\\hat{\\beta} + \\hat{\\epsilon},\n", + "\\label{_auto1} \\tag{1}\n", + "\\end{equation}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\hat{y}$ is a vector representing the possible outcomes, $\\hat{X}$ is our\n", + "$n\\times p$ design matrix and $\\hat{\\beta}$ represents our estimators/predictors.\n", + "\n", + "\n", + "The main problem with our function is that it takes values on the\n", + "entire real axis. In the case of logistic regression, however, the\n", + "labels $y_i$ are discrete variables. A typical example is the credit\n", + "card data discussed below here, where we can set the state of\n", + "defaulting the debt to $y_i=1$ and not to $y_i=0$ for one the persons\n", + "in the data set (see the full example below).\n", + "\n", + "One simple way to get a discrete output is to have sign\n", + "functions that map the output of a linear regressor to values $\\{0,1\\}$,\n", + "$f(s_i)=sign(s_i)=1$ if $s_i\\ge 0$ and 0 if otherwise. \n", + "We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron\" model in the machine learning\n", + "literature. This model is extremely simple. However, in many cases it is more\n", + "favorable to use a ``soft\" classifier that outputs\n", + "the probability of a given category. This leads us to the logistic function.\n", + "\n", + "\n", + "The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful." + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [ + { + "ename": "FileNotFoundError", + "evalue": "[Errno 2] No such file or directory: 'DataFiles/chddata.csv'", + "output_type": "error", + "traceback": [ + "\u001b[0;31m---------------------------------------------------------------------------\u001b[0m", + "\u001b[0;31mFileNotFoundError\u001b[0m Traceback (most recent call last)", + "\u001b[0;32m\u001b[0m in \u001b[0;36m\u001b[0;34m\u001b[0m\n\u001b[1;32m 38\u001b[0m \u001b[0mplt\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0msavefig\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mimage_path\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mfig_id\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;34m+\u001b[0m \u001b[0;34m\".png\"\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mformat\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m'png'\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 39\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 40\u001b[0;31m \u001b[0minfile\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mopen\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mdata_path\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m\"chddata.csv\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m'r'\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m 41\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 42\u001b[0m \u001b[0;31m# Read the chd data as csv file and organize the data into arrays with age group, age, and chd\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n", + "\u001b[0;31mFileNotFoundError\u001b[0m: [Errno 2] No such file or directory: 'DataFiles/chddata.csv'" + ] + } + ], + "source": [ + "%matplotlib inline\n", + "\n", + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.utils import resample\n", + "from sklearn.metrics import mean_squared_error\n", + "from IPython.display import display\n", + "from pylab import plt, mpl\n", + "plt.style.use('seaborn')\n", + "mpl.rcParams['font.family'] = 'serif'\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"chddata.csv\"),'r')\n", + "\n", + "# Read the chd data as csv file and organize the data into arrays with age group, age, and chd\n", + "chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))\n", + "chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']\n", + "output = chd['CHD']\n", + "age = chd['Age']\n", + "agegroup = chd['Agegroup']\n", + "numberID = chd['ID'] \n", + "display(chd)\n", + "\n", + "plt.scatter(age, output, marker='o')\n", + "plt.axis([18,70.0,-0.1, 1.2])\n", + "plt.xlabel(r'Age')\n", + "plt.ylabel(r'CHD')\n", + "plt.title(r'Age distribution and Coronary heart disease')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "What we could attempt however is to plot the mean value for each group." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])\n", + "group = np.array([1, 2, 3, 4, 5, 6, 7, 8])\n", + "plt.plot(group, agegroupmean, \"r-\")\n", + "plt.axis([0,9,0, 1.0])\n", + "plt.xlabel(r'Age group')\n", + "plt.ylabel(r'CHD mean values')\n", + "plt.title(r'Mean values for each age group')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We are now trying to find a function $f(y\\vert x)$, that is a function which gives us an expected value for the output $y$ with a given input $x$.\n", + "In standard linear regression with a linear dependence on $x$, we would write this in terms of our model" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(y_i\\vert x_i)=\\beta_0+\\beta_1 x_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This expression implies however that $f(y_i\\vert x_i)$ could take any\n", + "value from minus infinity to plus infinity. If we however let\n", + "$f(y\\vert y)$ be represented by the mean value, the above example\n", + "shows us that we can constrain the function to take values between\n", + "zero and one, that is we have $0 \\le f(y_i\\vert x_i) \\le 1$. Looking\n", + "at our last curve we see also that it has an S-shaped form. This leads\n", + "us to a very popular model for the function $f$, namely the so-called\n", + "Sigmoid function or logistic model. We will consider this function as\n", + "representing the probability for finding a value of $y_i$ with a given\n", + "$x_i$.\n", + "\n", + "\n", + "## The logistic function\n", + "\n", + "Another widely studied model, is the so-called \n", + "perceptron model, which is an example of a \"hard classification\" model. We\n", + "will encounter this model when we discuss neural networks as\n", + "well. Each datapoint is deterministically assigned to a category (i.e\n", + "$y_i=0$ or $y_i=1$). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a \"soft\"\n", + "classifier that outputs the probability of a given category rather\n", + "than a single value. For example, given $x_i$, the classifier\n", + "outputs the probability of being in a category $k$. Logistic regression\n", + "is the most common example of a so-called soft classifier. In logistic\n", + "regression, the probability that a data point $x_i$\n", + "belongs to a category $y_i=\\{0,1\\}$ is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(t) = \\frac{1}{1+\\mathrm \\exp{-t}}=\\frac{\\exp{t}}{1+\\mathrm \\exp{t}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Note that $1-p(t)= p(-t)$.\n", + "\n", + "## Examples of likelihood functions used in logistic regression and nueral networks\n", + "\n", + "\n", + "The following code plots the logistic function, the step function and other functions we will encounter from here and on." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\"\"\"The sigmoid function (or the logistic curve) is a\n", + "function that takes any real number, z, and outputs a number (0,1).\n", + "It is useful in neural networks for assigning weights on a relative scale.\n", + "The value z is the weighted sum of parameters involved in the learning algorithm.\"\"\"\n", + "\n", + "import numpy\n", + "import matplotlib.pyplot as plt\n", + "import math as mt\n", + "\n", + "z = numpy.arange(-5, 5, .1)\n", + "sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))\n", + "sigma = sigma_fn(z)\n", + "\n", + "fig = plt.figure()\n", + "ax = fig.add_subplot(111)\n", + "ax.plot(z, sigma)\n", + "ax.set_ylim([-0.1, 1.1])\n", + "ax.set_xlim([-5,5])\n", + "ax.grid(True)\n", + "ax.set_xlabel('z')\n", + "ax.set_title('sigmoid function')\n", + "\n", + "plt.show()\n", + "\n", + "\"\"\"Step Function\"\"\"\n", + "z = numpy.arange(-5, 5, .02)\n", + "step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)\n", + "step = step_fn(z)\n", + "\n", + "fig = plt.figure()\n", + "ax = fig.add_subplot(111)\n", + "ax.plot(z, step)\n", + "ax.set_ylim([-0.5, 1.5])\n", + "ax.set_xlim([-5,5])\n", + "ax.grid(True)\n", + "ax.set_xlabel('z')\n", + "ax.set_title('step function')\n", + "\n", + "plt.show()\n", + "\n", + "\"\"\"tanh Function\"\"\"\n", + "z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)\n", + "t = numpy.tanh(z)\n", + "\n", + "fig = plt.figure()\n", + "ax = fig.add_subplot(111)\n", + "ax.plot(z, t)\n", + "ax.set_ylim([-1.0, 1.0])\n", + "ax.set_xlim([-2*mt.pi,2*mt.pi])\n", + "ax.grid(True)\n", + "ax.set_xlabel('z')\n", + "ax.set_title('tanh function')\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We assume now that we have two classes with $y_i$ either $0$ or $1$. Furthermore we assume also that we have only two parameters $\\beta$ in our fitting of the Sigmoid function, that is we define probabilities" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "p(y_i=1|x_i,\\hat{\\beta}) &= \\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}},\\nonumber\\\\\n", + "p(y_i=0|x_i,\\hat{\\beta}) &= 1 - p(y_i=1|x_i,\\hat{\\beta}),\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\hat{\\beta}$ are the weights we wish to extract from data, in our case $\\beta_0$ and $\\beta_1$. \n", + "\n", + "Note that we used" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(y_i=0\\vert x_i, \\hat{\\beta}) = 1-p(y_i=1\\vert x_i, \\hat{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In order to define the total likelihood for all possible outcomes from a \n", + "dataset $\\mathcal{D}=\\{(y_i,x_i)\\}$, with the binary labels\n", + "$y_i\\in\\{0,1\\}$ and where the data points are drawn independently, we use the so-called [Maximum Likelihood Estimation](https://en.wikipedia.org/wiki/Maximum_likelihood_estimation) (MLE) principle. \n", + "We aim thus at maximizing \n", + "the probability of seeing the observed data. We can then approximate the \n", + "likelihood in terms of the product of the individual probabilities of a specific outcome $y_i$, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "P(\\mathcal{D}|\\hat{\\beta})& = \\prod_{i=1}^n \\left[p(y_i=1|x_i,\\hat{\\beta})\\right]^{y_i}\\left[1-p(y_i=1|x_i,\\hat{\\beta}))\\right]^{1-y_i}\\nonumber \\\\\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "from which we obtain the log-likelihood and our **cost/loss** function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathcal{C}(\\hat{\\beta}) = \\sum_{i=1}^n \\left( y_i\\log{p(y_i=1|x_i,\\hat{\\beta})} + (1-y_i)\\log\\left[1-p(y_i=1|x_i,\\hat{\\beta}))\\right]\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Reordering the logarithms, we can rewrite the **cost/loss** function as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathcal{C}(\\hat{\\beta}) = \\sum_{i=1}^n \\left(y_i(\\beta_0+\\beta_1x_i) -\\log{(1+\\exp{(\\beta_0+\\beta_1x_i)})}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to $\\beta$.\n", + "Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathcal{C}(\\hat{\\beta})=-\\sum_{i=1}^n \\left(y_i(\\beta_0+\\beta_1x_i) -\\log{(1+\\exp{(\\beta_0+\\beta_1x_i)})}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This equation is known in statistics as the **cross entropy**. Finally, we note that just as in linear regression, \n", + "in practice we often supplement the cross-entropy with additional regularization terms, usually $L_1$ and $L_2$ regularization as we did for Ridge and Lasso regression.\n", + "\n", + "\n", + "The cross entropy is a convex function of the weights $\\hat{\\beta}$ and,\n", + "therefore, any local minimizer is a global minimizer. \n", + "\n", + "\n", + "Minimizing this\n", + "cost function with respect to the two parameters $\\beta_0$ and $\\beta_1$ we obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\hat{\\beta})}{\\partial \\beta_0} = -\\sum_{i=1}^n \\left(y_i -\\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}}\\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\hat{\\beta})}{\\partial \\beta_1} = -\\sum_{i=1}^n \\left(y_ix_i -x_i\\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Let us now define a vector $\\hat{y}$ with $n$ elements $y_i$, an\n", + "$n\\times p$ matrix $\\hat{X}$ which contains the $x_i$ values and a\n", + "vector $\\hat{p}$ of fitted probabilities $p(y_i\\vert x_i,\\hat{\\beta})$. We can rewrite in a more compact form the first\n", + "derivative of cost function as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\hat{\\beta})}{\\partial \\hat{\\beta}} = -\\hat{X}^T\\left(\\hat{y}-\\hat{p}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we in addition define a diagonal matrix $\\hat{W}$ with elements \n", + "$p(y_i\\vert x_i,\\hat{\\beta})(1-p(y_i\\vert x_i,\\hat{\\beta})$, we can obtain a compact expression of the second derivative as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial^2 \\mathcal{C}(\\hat{\\beta})}{\\partial \\hat{\\beta}\\partial \\hat{\\beta}^T} = \\hat{X}^T\\hat{W}\\hat{X}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with $p$ predictors" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{ \\frac{p(\\hat{\\beta}\\hat{x})}{1-p(\\hat{\\beta}\\hat{x})}} = \\beta_0+\\beta_1x_1+\\beta_2x_2+\\dots+\\beta_px_p.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here we defined $\\hat{x}=[1,x_1,x_2,\\dots,x_p]$ and $\\hat{\\beta}=[\\beta_0, \\beta_1, \\dots, \\beta_p]$ leading to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(\\hat{\\beta}\\hat{x})=\\frac{ \\exp{(\\beta_0+\\beta_1x_1+\\beta_2x_2+\\dots+\\beta_px_p)}}{1+\\exp{(\\beta_0+\\beta_1x_1+\\beta_2x_2+\\dots+\\beta_px_p)}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Till now we have mainly focused on two classes, the so-called binary\n", + "system. Suppose we wish to extend to $K$ classes. Let us for the sake\n", + "of simplicity assume we have only two predictors. We have then following model" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{\\frac{p(C=1\\vert x)}{p(K\\vert x)}} = \\beta_{10}+\\beta_{11}x_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{\\frac{p(C=2\\vert x)}{p(K\\vert x)}} = \\beta_{20}+\\beta_{21}x_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and so on till the class $C=K-1$ class" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{\\frac{p(C=K-1\\vert x)}{p(K\\vert x)}} = \\beta_{(K-1)0}+\\beta_{(K-1)1}x_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and the model is specified in term of $K-1$ so-called log-odds or\n", + "**logit** transformations.\n", + "\n", + "\n", + "\n", + "In our discussion of neural networks we will encounter the above again\n", + "in terms of a slightly modified function, the so-called **Softmax** function.\n", + "\n", + "The softmax function is used in various multiclass classification\n", + "methods, such as multinomial logistic regression (also known as\n", + "softmax regression), multiclass linear discriminant analysis, naive\n", + "Bayes classifiers, and artificial neural networks. Specifically, in\n", + "multinomial logistic regression and linear discriminant analysis, the\n", + "input to the function is the result of $K$ distinct linear functions,\n", + "and the predicted probability for the $k$-th class given a sample\n", + "vector $\\hat{x}$ and a weighting vector $\\hat{\\beta}$ is (with two\n", + "predictors):" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(C=k\\vert \\mathbf {x} )=\\frac{\\exp{(\\beta_{k0}+\\beta_{k1}x_1)}}{1+\\sum_{l=1}^{K-1}\\exp{(\\beta_{l0}+\\beta_{l1}x_1)}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "It is easy to extend to more predictors. The final class is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(C=K\\vert \\mathbf {x} )=\\frac{1}{1+\\sum_{l=1}^{K-1}\\exp{(\\beta_{l0}+\\beta_{l1}x_1)}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and they sum to one. Our earlier discussions were all specialized to\n", + "the case with two classes only. It is easy to see from the above that\n", + "what we derived earlier is compatible with these equations.\n", + "\n", + "To find the optimal parameters we would typically use a gradient\n", + "descent method. Newton's method and gradient descent methods are\n", + "discussed in the material on [optimization\n", + "methods](https://compphysics.github.io/MachineLearning/doc/pub/Splines/html/Splines-bs.html).\n", + "\n", + "## Wisconsin Cancer Data\n", + "\n", + "We show here how we can use a simple regression case on the breast\n", + "cancer data using Logistic regression as our algorithm for\n", + "classification." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.linear_model import LogisticRegression\n", + "\n", + "# Load the data\n", + "cancer = load_breast_cancer()\n", + "\n", + "X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)\n", + "print(X_train.shape)\n", + "print(X_test.shape)\n", + "# Logistic Regression\n", + "logreg = LogisticRegression(solver='lbfgs')\n", + "logreg.fit(X_train, y_train)\n", + "print(\"Test set accuracy with Logistic Regression: {:.2f}\".format(logreg.score(X_test,y_test)))\n", + "#now scale the data\n", + "from sklearn.preprocessing import StandardScaler\n", + "scaler = StandardScaler()\n", + "scaler.fit(X_train)\n", + "X_train_scaled = scaler.transform(X_train)\n", + "X_test_scaled = scaler.transform(X_test)\n", + "# Logistic Regression\n", + "logreg.fit(X_train_scaled, y_train)\n", + "print(\"Test set accuracy Logistic Regression with scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In addition to the above scores, we could also study the covariance (and the correlation matrix).\n", + "We use **Pandas** to compute the correlation matrix." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.linear_model import LogisticRegression\n", + "cancer = load_breast_cancer()\n", + "import pandas as pd\n", + "# Making a data frame\n", + "cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)\n", + "\n", + "fig, axes = plt.subplots(15,2,figsize=(10,20))\n", + "malignant = cancer.data[cancer.target == 0]\n", + "benign = cancer.data[cancer.target == 1]\n", + "ax = axes.ravel()\n", + "\n", + "for i in range(30):\n", + " _, bins = np.histogram(cancer.data[:,i], bins =50)\n", + " ax[i].hist(malignant[:,i], bins = bins, alpha = 0.5)\n", + " ax[i].hist(benign[:,i], bins = bins, alpha = 0.5)\n", + " ax[i].set_title(cancer.feature_names[i])\n", + " ax[i].set_yticks(())\n", + "ax[0].set_xlabel(\"Feature magnitude\")\n", + "ax[0].set_ylabel(\"Frequency\")\n", + "ax[0].legend([\"Malignant\", \"Benign\"], loc =\"best\")\n", + "fig.tight_layout()\n", + "plt.show()\n", + "\n", + "import seaborn as sns\n", + "correlation_matrix = cancerpd.corr().round(1)\n", + "# use the heatmap function from seaborn to plot the correlation matrix\n", + "# annot = True to print the values inside the square\n", + "plt.figure(figsize=(15,8))\n", + "sns.heatmap(data=correlation_matrix, annot=True)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the above example we note two things. In the first plot we display\n", + "the overlap of benign and malignant tumors as functions of the various\n", + "features in the Wisconsing breast cancer data set. We see that for\n", + "some of the features we can distinguish clearly the benign and\n", + "malignant cases while for other features we cannot. This can point to\n", + "us which features may be of greater interest when we wish to classify\n", + "a benign or not benign tumour.\n", + "\n", + "In the second figure we have computed the so-called correlation\n", + "matrix, which in our case with thirty features becomes a $30\\times 30$\n", + "matrix.\n", + "\n", + "We constructed this matrix using **pandas** via the statements" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and then" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "correlation_matrix = cancerpd.corr().round(1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Diagonalizing this matrix we can in turn say something about which\n", + "features are of relevance and which are not. This leads us to\n", + "the classical Principal Component Analysis (PCA) theorem with\n", + "applications. This will be discussed later this semester ([week 43](https://compphysics.github.io/MachineLearning/doc/pub/week43/html/week43-bs.html))." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.linear_model import LogisticRegression\n", + "\n", + "# Load the data\n", + "cancer = load_breast_cancer()\n", + "\n", + "X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)\n", + "print(X_train.shape)\n", + "print(X_test.shape)\n", + "# Logistic Regression\n", + "logreg = LogisticRegression(solver='lbfgs')\n", + "logreg.fit(X_train, y_train)\n", + "print(\"Test set accuracy with Logistic Regression: {:.2f}\".format(logreg.score(X_test,y_test)))\n", + "#now scale the data\n", + "from sklearn.preprocessing import StandardScaler\n", + "scaler = StandardScaler()\n", + "scaler.fit(X_train)\n", + "X_train_scaled = scaler.transform(X_train)\n", + "X_test_scaled = scaler.transform(X_test)\n", + "# Logistic Regression\n", + "logreg.fit(X_train_scaled, y_train)\n", + "print(\"Test set accuracy Logistic Regression with scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))\n", + "\n", + "\n", + "from sklearn.preprocessing import LabelEncoder\n", + "from sklearn.model_selection import cross_validate\n", + "#Cross validation\n", + "accuracy = cross_validate(logreg,X_test_scaled,y_test,cv=10)['test_score']\n", + "print(accuracy)\n", + "print(\"Test set accuracy with Logistic Regression and scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))\n", + "\n", + "\n", + "import scikitplot as skplt\n", + "y_pred = logreg.predict(X_test_scaled)\n", + "skplt.metrics.plot_confusion_matrix(y_test, y_pred, normalize=True)\n", + "plt.show()\n", + "y_probas = logreg.predict_proba(X_test_scaled)\n", + "skplt.metrics.plot_roc(y_test, y_probas)\n", + "plt.show()\n", + "skplt.metrics.plot_cumulative_gain(y_test, y_probas)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Optimization, the central part of any Machine Learning algortithm\n", + "\n", + "Almost every problem in machine learning and data science starts with\n", + "a dataset $X$, a model $g(\\beta)$, which is a function of the\n", + "parameters $\\beta$ and a cost function $C(X, g(\\beta))$ that allows\n", + "us to judge how well the model $g(\\beta)$ explains the observations\n", + "$X$. The model is fit by finding the values of $\\beta$ that minimize\n", + "the cost function. Ideally we would be able to solve for $\\beta$\n", + "analytically, however this is not possible in general and we must use\n", + "some approximative/numerical method to compute the minimum.\n", + "\n", + "\n", + "\n", + "## Revisiting our Logistic Regression case\n", + "\n", + "In our discussion on Logistic Regression we studied the \n", + "case of\n", + "two classes, with $y_i$ either\n", + "$0$ or $1$. Furthermore we assumed also that we have only two\n", + "parameters $\\beta$ in our fitting, that is we\n", + "defined probabilities" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "p(y_i=1|x_i,\\boldsymbol{\\beta}) &= \\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}},\\nonumber\\\\\n", + "p(y_i=0|x_i,\\boldsymbol{\\beta}) &= 1 - p(y_i=1|x_i,\\boldsymbol{\\beta}),\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\boldsymbol{\\beta}$ are the weights we wish to extract from data, in our case $\\beta_0$ and $\\beta_1$. \n", + "\n", + "\n", + "## The equations to solve\n", + "\n", + "Our compact equations used a definition of a vector $\\boldsymbol{y}$ with $n$\n", + "elements $y_i$, an $n\\times p$ matrix $\\boldsymbol{X}$ which contains the\n", + "$x_i$ values and a vector $\\boldsymbol{p}$ of fitted probabilities\n", + "$p(y_i\\vert x_i,\\boldsymbol{\\beta})$. We rewrote in a more compact form\n", + "the first derivative of the cost function as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = -\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{p}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we in addition define a diagonal matrix $\\boldsymbol{W}$ with elements \n", + "$p(y_i\\vert x_i,\\boldsymbol{\\beta})(1-p(y_i\\vert x_i,\\boldsymbol{\\beta})$, we can obtain a compact expression of the second derivative as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial^2 \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}\\partial \\boldsymbol{\\beta}^T} = \\boldsymbol{X}^T\\boldsymbol{W}\\boldsymbol{X}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This defines what is called the Hessian matrix.\n", + "\n", + "\n", + "## Solving using Newton-Raphson's method\n", + "\n", + "If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives. \n", + "\n", + "Our iterative scheme is then given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{new}} = \\boldsymbol{\\beta}^{\\mathrm{old}}-\\left(\\frac{\\partial^2 \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}\\partial \\boldsymbol{\\beta}^T}\\right)^{-1}_{\\boldsymbol{\\beta}^{\\mathrm{old}}}\\times \\left(\\frac{\\partial \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}}\\right)_{\\boldsymbol{\\beta}^{\\mathrm{old}}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or in matrix form as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{new}} = \\boldsymbol{\\beta}^{\\mathrm{old}}-\\left(\\boldsymbol{X}^T\\boldsymbol{W}\\boldsymbol{X} \\right)^{-1}\\times \\left(-\\boldsymbol{X}^T(\\boldsymbol{y}-\\boldsymbol{p}) \\right)_{\\boldsymbol{\\beta}^{\\mathrm{old}}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The right-hand side is computed with the old values of $\\beta$. \n", + "\n", + "If we can compute these matrices, in particular the Hessian, the above is often the easiest method to implement. \n", + "\n", + "\n", + "\n", + "## Brief reminder on Newton-Raphson's method\n", + "\n", + "Let us quickly remind ourselves how we derive the above method.\n", + "\n", + "Perhaps the most celebrated of all one-dimensional root-finding\n", + "routines is Newton's method, also called the Newton-Raphson\n", + "method. This method requires the evaluation of both the\n", + "function $f$ and its derivative $f'$ at arbitrary points. \n", + "If you can only calculate the derivative\n", + "numerically and/or your function is not of the smooth type, we\n", + "normally discourage the use of this method.\n", + "\n", + "\n", + "## The equations\n", + "\n", + "The Newton-Raphson formula consists geometrically of extending the\n", + "tangent line at a current point until it crosses zero, then setting\n", + "the next guess to the abscissa of that zero-crossing. The mathematics\n", + "behind this method is rather simple. Employing a Taylor expansion for\n", + "$x$ sufficiently close to the solution $s$, we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "
\n", + "\n", + "$$\n", + "f(s)=0=f(x)+(s-x)f'(x)+\\frac{(s-x)^2}{2}f''(x) +\\dots.\n", + " \\label{eq:taylornr} \\tag{2}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "For small enough values of the function and for well-behaved\n", + "functions, the terms beyond linear are unimportant, hence we obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(x)+(s-x)f'(x)\\approx 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "yielding" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "s\\approx x-\\frac{f(x)}{f'(x)}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Having in mind an iterative procedure, it is natural to start iterating with" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "x_{n+1}=x_n-\\frac{f(x_n)}{f'(x_n)}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Simple geometric interpretation\n", + "\n", + "The above is Newton-Raphson's method. It has a simple geometric\n", + "interpretation, namely $x_{n+1}$ is the point where the tangent from\n", + "$(x_n,f(x_n))$ crosses the $x$-axis. Close to the solution,\n", + "Newton-Raphson converges fast to the desired result. However, if we\n", + "are far from a root, where the higher-order terms in the series are\n", + "important, the Newton-Raphson formula can give grossly inaccurate\n", + "results. For instance, the initial guess for the root might be so far\n", + "from the true root as to let the search interval include a local\n", + "maximum or minimum of the function. If an iteration places a trial\n", + "guess near such a local extremum, so that the first derivative nearly\n", + "vanishes, then Newton-Raphson may fail totally\n", + "\n", + "\n", + "\n", + "## Extending to more than one variable\n", + "\n", + "Newton's method can be generalized to systems of several non-linear equations\n", + "and variables. Consider the case with two equations" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{array}{cc} f_1(x_1,x_2) &=0\\\\\n", + " f_2(x_1,x_2) &=0,\\end{array}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which we Taylor expand to obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{array}{cc} 0=f_1(x_1+h_1,x_2+h_2)=&f_1(x_1,x_2)+h_1\n", + " \\partial f_1/\\partial x_1+h_2\n", + " \\partial f_1/\\partial x_2+\\dots\\\\\n", + " 0=f_2(x_1+h_1,x_2+h_2)=&f_2(x_1,x_2)+h_1\n", + " \\partial f_2/\\partial x_1+h_2\n", + " \\partial f_2/\\partial x_2+\\dots\n", + " \\end{array}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Defining the Jacobian matrix ${\\bf \\boldsymbol{J}}$ we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\bf \\boldsymbol{J}}=\\left( \\begin{array}{cc}\n", + " \\partial f_1/\\partial x_1 & \\partial f_1/\\partial x_2 \\\\\n", + " \\partial f_2/\\partial x_1 &\\partial f_2/\\partial x_2\n", + " \\end{array} \\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we can rephrase Newton's method as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\left(\\begin{array}{c} x_1^{n+1} \\\\ x_2^{n+1} \\end{array} \\right)=\n", + "\\left(\\begin{array}{c} x_1^{n} \\\\ x_2^{n} \\end{array} \\right)+\n", + "\\left(\\begin{array}{c} h_1^{n} \\\\ h_2^{n} \\end{array} \\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have defined" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\left(\\begin{array}{c} h_1^{n} \\\\ h_2^{n} \\end{array} \\right)=\n", + " -{\\bf \\boldsymbol{J}}^{-1}\n", + " \\left(\\begin{array}{c} f_1(x_1^{n},x_2^{n}) \\\\ f_2(x_1^{n},x_2^{n}) \\end{array} \\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We need thus to compute the inverse of the Jacobian matrix and it\n", + "is to understand that difficulties may\n", + "arise in case ${\\bf \\boldsymbol{J}}$ is nearly singular.\n", + "\n", + "It is rather straightforward to extend the above scheme to systems of\n", + "more than two non-linear equations. In our case, the Jacobian matrix is given by the Hessian that represents the second derivative of cost function. \n", + "\n", + "\n", + "\n", + "\n", + "## Steepest descent\n", + "\n", + "The basic idea of gradient descent is\n", + "that a function $F(\\mathbf{x})$, \n", + "$\\mathbf{x} \\equiv (x_1,\\cdots,x_n)$, decreases fastest if one goes from $\\bf {x}$ in the\n", + "direction of the negative gradient $-\\nabla F(\\mathbf{x})$.\n", + "\n", + "It can be shown that if" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{x}_{k+1} = \\mathbf{x}_k - \\gamma_k \\nabla F(\\mathbf{x}_k),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\gamma_k > 0$.\n", + "\n", + "For $\\gamma_k$ small enough, then $F(\\mathbf{x}_{k+1}) \\leq\n", + "F(\\mathbf{x}_k)$. This means that for a sufficiently small $\\gamma_k$\n", + "we are always moving towards smaller function values, i.e a minimum.\n", + "\n", + "\n", + "## More on Steepest descent\n", + "\n", + "The previous observation is the basis of the method of steepest\n", + "descent, which is also referred to as just gradient descent (GD). One\n", + "starts with an initial guess $\\mathbf{x}_0$ for a minimum of $F$ and\n", + "computes new approximations according to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{x}_{k+1} = \\mathbf{x}_k - \\gamma_k \\nabla F(\\mathbf{x}_k), \\ \\ k \\geq 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The parameter $\\gamma_k$ is often referred to as the step length or\n", + "the learning rate within the context of Machine Learning.\n", + "\n", + "\n", + "## The ideal\n", + "\n", + "Ideally the sequence $\\{\\mathbf{x}_k \\}_{k=0}$ converges to a global\n", + "minimum of the function $F$. In general we do not know if we are in a\n", + "global or local minimum. In the special case when $F$ is a convex\n", + "function, all local minima are also global minima, so in this case\n", + "gradient descent can converge to the global solution. The advantage of\n", + "this scheme is that it is conceptually simple and straightforward to\n", + "implement. However the method in this form has some severe\n", + "limitations:\n", + "\n", + "In machine learing we are often faced with non-convex high dimensional\n", + "cost functions with many local minima. Since GD is deterministic we\n", + "will get stuck in a local minimum, if the method converges, unless we\n", + "have a very good intial guess. This also implies that the scheme is\n", + "sensitive to the chosen initial condition.\n", + "\n", + "Note that the gradient is a function of $\\mathbf{x} =\n", + "(x_1,\\cdots,x_n)$ which makes it expensive to compute numerically.\n", + "\n", + "\n", + "\n", + "## The sensitiveness of the gradient descent\n", + "\n", + "The gradient descent method \n", + "is sensitive to the choice of learning rate $\\gamma_k$. This is due\n", + "to the fact that we are only guaranteed that $F(\\mathbf{x}_{k+1}) \\leq\n", + "F(\\mathbf{x}_k)$ for sufficiently small $\\gamma_k$. The problem is to\n", + "determine an optimal learning rate. If the learning rate is chosen too\n", + "small the method will take a long time to converge and if it is too\n", + "large we can experience erratic behavior.\n", + "\n", + "Many of these shortcomings can be alleviated by introducing\n", + "randomness. One such method is that of Stochastic Gradient Descent\n", + "(SGD), see below.\n", + "\n", + "\n", + "\n", + "## Convex functions\n", + "\n", + "Ideally we want our cost/loss function to be convex(concave).\n", + "\n", + "First we give the definition of a convex set: A set $C$ in\n", + "$\\mathbb{R}^n$ is said to be convex if, for all $x$ and $y$ in $C$ and\n", + "all $t \\in (0,1)$ , the point $(1 − t)x + ty$ also belongs to\n", + "C. Geometrically this means that every point on the line segment\n", + "connecting $x$ and $y$ is in $C$ as discussed below.\n", + "\n", + "The convex subsets of $\\mathbb{R}$ are the intervals of\n", + "$\\mathbb{R}$. Examples of convex sets of $\\mathbb{R}^2$ are the\n", + "regular polygons (triangles, rectangles, pentagons, etc...).\n", + "\n", + "\n", + "## Convex function\n", + "\n", + "**Convex function**: Let $X \\subset \\mathbb{R}^n$ be a convex set. Assume that the function $f: X \\rightarrow \\mathbb{R}$ is continuous, then $f$ is said to be convex if $$f(tx_1 + (1-t)x_2) \\leq tf(x_1) + (1-t)f(x_2) $$ for all $x_1, x_2 \\in X$ and for all $t \\in [0,1]$. If $\\leq$ is replaced with a strict inequaltiy in the definition, we demand $x_1 \\neq x_2$ and $t\\in(0,1)$ then $f$ is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting $f(x_1)$ and $f(x_2)$, the value of the function on the interval $[x_1,x_2]$ is always below the line as illustrated below.\n", + "\n", + "\n", + "## Conditions on convex functions\n", + "\n", + "In the following we state first and second-order conditions which\n", + "ensures convexity of a function $f$. We write $D_f$ to denote the\n", + "domain of $f$, i.e the subset of $R^n$ where $f$ is defined. For more\n", + "details and proofs we refer to: [S. Boyd and L. Vandenberghe. Convex Optimization. Cambridge University Press](http://stanford.edu/boyd/cvxbook/, 2004).\n", + "\n", + "**First order condition.**\n", + "\n", + "Suppose $f$ is differentiable (i.e $\\nabla f(x)$ is well defined for\n", + "all $x$ in the domain of $f$). Then $f$ is convex if and only if $D_f$\n", + "is a convex set and $$f(y) \\geq f(x) + \\nabla f(x)^T (y-x) $$ holds\n", + "for all $x,y \\in D_f$. This condition means that for a convex function\n", + "the first order Taylor expansion (right hand side above) at any point\n", + "a global under estimator of the function. To convince yourself you can\n", + "make a drawing of $f(x) = x^2+1$ and draw the tangent line to $f(x)$ and\n", + "note that it is always below the graph.\n", + "\n", + "\n", + "\n", + "**Second order condition.**\n", + "\n", + "Assume that $f$ is twice\n", + "differentiable, i.e the Hessian matrix exists at each point in\n", + "$D_f$. Then $f$ is convex if and only if $D_f$ is a convex set and its\n", + "Hessian is positive semi-definite for all $x\\in D_f$. For a\n", + "single-variable function this reduces to $f''(x) \\geq 0$. Geometrically this means that $f$ has nonnegative curvature\n", + "everywhere.\n", + "\n", + "\n", + "\n", + "This condition is particularly useful since it gives us an procedure for determining if the function under consideration is convex, apart from using the definition.\n", + "\n", + "\n", + "## More on convex functions\n", + "\n", + "The next result is of great importance to us and the reason why we are\n", + "going on about convex functions. In machine learning we frequently\n", + "have to minimize a loss/cost function in order to find the best\n", + "parameters for the model we are considering. \n", + "\n", + "Ideally we want the\n", + "global minimum (for high-dimensional models it is hard to know\n", + "if we have local or global minimum). However, if the cost/loss function\n", + "is convex the following result provides invaluable information:\n", + "\n", + "**Any minimum is global for convex functions.**\n", + "\n", + "Consider the problem of finding $x \\in \\mathbb{R}^n$ such that $f(x)$\n", + "is minimal, where $f$ is convex and differentiable. Then, any point\n", + "$x^*$ that satisfies $\\nabla f(x^*) = 0$ is a global minimum.\n", + "\n", + "\n", + "\n", + "This result means that if we know that the cost/loss function is convex and we are able to find a minimum, we are guaranteed that it is a global minimum.\n", + "\n", + "\n", + "## Some simple problems\n", + "\n", + "1. Show that $f(x)=x^2$ is convex for $x \\in \\mathbb{R}$ using the definition of convexity. Hint: If you re-write the definition, $f$ is convex if the following holds for all $x,y \\in D_f$ and any $\\lambda \\in [0,1]$ $\\lambda f(x)+(1-\\lambda)f(y)-f(\\lambda x + (1-\\lambda) y ) \\geq 0$.\n", + "\n", + "2. Using the second order condition show that the following functions are convex on the specified domain.\n", + "\n", + " * $f(x) = e^x$ is convex for $x \\in \\mathbb{R}$.\n", + "\n", + " * $g(x) = -\\ln(x)$ is convex for $x \\in (0,\\infty)$.\n", + "\n", + "\n", + "3. Let $f(x) = x^2$ and $g(x) = e^x$. Show that $f(g(x))$ and $g(f(x))$ is convex for $x \\in \\mathbb{R}$. Also show that if $f(x)$ is any convex function than $h(x) = e^{f(x)}$ is convex.\n", + "\n", + "4. A norm is any function that satisfy the following properties\n", + "\n", + " * $f(\\alpha x) = |\\alpha| f(x)$ for all $\\alpha \\in \\mathbb{R}$.\n", + "\n", + " * $f(x+y) \\leq f(x) + f(y)$\n", + "\n", + " * $f(x) \\leq 0$ for all $x \\in \\mathbb{R}^n$ with equality if and only if $x = 0$\n", + "\n", + "\n", + "Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this).\n", + "\n", + "\n", + "\n", + "## Friday September 25\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember25.mp4?vrtx=view-as-webpage) and [link to handwritten notes](https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/NotesSeptember25.pdf).\n", + "\n", + "\n", + "\n", + "## Standard steepest descent\n", + "\n", + "\n", + "Before we proceed, we would like to discuss the approach called the\n", + "**standard Steepest descent** (different from the above steepest descent discussion), which again leads to us having to be able\n", + "to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG).\n", + "\n", + "[The success of the CG method](https://www.cs.cmu.edu/~quake-papers/painless-conjugate-gradient.pdf)\n", + "for finding solutions of non-linear problems is based on the theory\n", + "of conjugate gradients for linear systems of equations. It belongs to\n", + "the class of iterative methods for solving problems from linear\n", + "algebra of the type" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{x} = \\boldsymbol{b}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the iterative process we end up with a problem like" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}= \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\boldsymbol{r}$ is the so-called residual or error in the iterative process.\n", + "\n", + "When we have found the exact solution, $\\boldsymbol{r}=0$.\n", + "\n", + "\n", + "## Gradient method\n", + "\n", + "The residual is zero when we reach the minimum of the quadratic equation" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "P(\\boldsymbol{x})=\\frac{1}{2}\\boldsymbol{x}^T\\boldsymbol{A}\\boldsymbol{x} - \\boldsymbol{x}^T\\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with the constraint that the matrix $\\boldsymbol{A}$ is positive definite and\n", + "symmetric. This defines also the Hessian and we want it to be positive definite. \n", + "\n", + "\n", + "\n", + "## Steepest descent method\n", + "\n", + "We denote the initial guess for $\\boldsymbol{x}$ as $\\boldsymbol{x}_0$. \n", + "We can assume without loss of generality that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_0=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or consider the system" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{z} = \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "instead.\n", + "\n", + "\n", + "\n", + "## Steepest descent method\n", + "One can show that the solution $\\boldsymbol{x}$ is also the unique minimizer of the quadratic form" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(\\boldsymbol{x}) = \\frac{1}{2}\\boldsymbol{x}^T\\boldsymbol{A}\\boldsymbol{x} - \\boldsymbol{x}^T \\boldsymbol{x} , \\quad \\boldsymbol{x}\\in\\mathbf{R}^n.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This suggests taking the first basis vector $\\boldsymbol{r}_1$ (see below for definition) \n", + "to be the gradient of $f$ at $\\boldsymbol{x}=\\boldsymbol{x}_0$, \n", + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{x}_0-\\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and \n", + "$\\boldsymbol{x}_0=0$ it is equal $-\\boldsymbol{b}$.\n", + "\n", + "\n", + "\n", + "\n", + "## Final expressions\n", + "We can compute the residual iteratively as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_{k+1}=\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_{k+1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{b}-\\boldsymbol{A}(\\boldsymbol{x}_k+\\alpha_k\\boldsymbol{r}_k),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_k)-\\alpha_k\\boldsymbol{A}\\boldsymbol{r}_k,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which gives" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\alpha_k = \\frac{\\boldsymbol{r}_k^T\\boldsymbol{r}_k}{\\boldsymbol{r}_k^T\\boldsymbol{A}\\boldsymbol{r}_k}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "leading to the iterative scheme" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_{k+1}=\\boldsymbol{x}_k-\\alpha_k\\boldsymbol{r}_{k},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Steepest descent example" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import numpy.linalg as la\n", + "\n", + "import scipy.optimize as sopt\n", + "\n", + "import matplotlib.pyplot as pt\n", + "from mpl_toolkits.mplot3d import axes3d\n", + "\n", + "def f(x):\n", + " return 0.5*x[0]**2 + 2.5*x[1]**2\n", + "\n", + "def df(x):\n", + " return np.array([x[0], 5*x[1]])\n", + "\n", + "fig = pt.figure()\n", + "ax = fig.gca(projection=\"3d\")\n", + "\n", + "xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]\n", + "fmesh = f(np.array([xmesh, ymesh]))\n", + "ax.plot_surface(xmesh, ymesh, fmesh)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "And then as countor plot" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "pt.axis(\"equal\")\n", + "pt.contour(xmesh, ymesh, fmesh)\n", + "guesses = [np.array([2, 2./5])]" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Find guesses" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "x = guesses[-1]\n", + "s = -df(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Run it!" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "def f1d(alpha):\n", + " return f(x + alpha*s)\n", + "\n", + "alpha_opt = sopt.golden(f1d)\n", + "next_guess = x + alpha_opt * s\n", + "guesses.append(next_guess)\n", + "print(next_guess)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "What happened?" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "pt.axis(\"equal\")\n", + "pt.contour(xmesh, ymesh, fmesh, 50)\n", + "it_array = np.array(guesses)\n", + "pt.plot(it_array.T[0], it_array.T[1], \"x-\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method\n", + "In the CG method we define so-called conjugate directions and two vectors \n", + "$\\boldsymbol{s}$ and $\\boldsymbol{t}$\n", + "are said to be\n", + "conjugate if" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{s}^T\\boldsymbol{A}\\boldsymbol{t}= 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The philosophy of the CG method is to perform searches in various conjugate directions\n", + "of our vectors $\\boldsymbol{x}_i$ obeying the above criterion, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_i^T\\boldsymbol{A}\\boldsymbol{x}_j= 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Two vectors are conjugate if they are orthogonal with respect to \n", + "this inner product. Being conjugate is a symmetric relation: if $\\boldsymbol{s}$ is conjugate to $\\boldsymbol{t}$, then $\\boldsymbol{t}$ is conjugate to $\\boldsymbol{s}$.\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "An example is given by the eigenvectors of the matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{v}_i^T\\boldsymbol{A}\\boldsymbol{v}_j= \\lambda\\boldsymbol{v}_i^T\\boldsymbol{v}_j,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which is zero unless $i=j$.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "Assume now that we have a symmetric positive-definite matrix $\\boldsymbol{A}$ of size\n", + "$n\\times n$. At each iteration $i+1$ we obtain the conjugate direction of a vector" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_{i+1}=\\boldsymbol{x}_{i}+\\alpha_i\\boldsymbol{p}_{i}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We assume that $\\boldsymbol{p}_{i}$ is a sequence of $n$ mutually conjugate directions. \n", + "Then the $\\boldsymbol{p}_{i}$ form a basis of $R^n$ and we can expand the solution \n", + "$ \\boldsymbol{A}\\boldsymbol{x} = \\boldsymbol{b}$ in this basis, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x} = \\sum^{n}_{i=1} \\alpha_i \\boldsymbol{p}_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method\n", + "The coefficients are given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{A}\\mathbf{x} = \\sum^{n}_{i=1} \\alpha_i \\mathbf{A} \\mathbf{p}_i = \\mathbf{b}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Multiplying with $\\boldsymbol{p}_k^T$ from the left gives" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{p}_k^T \\boldsymbol{A}\\boldsymbol{x} = \\sum^{n}_{i=1} \\alpha_i\\boldsymbol{p}_k^T \\boldsymbol{A}\\boldsymbol{p}_i= \\boldsymbol{p}_k^T \\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and we can define the coefficients $\\alpha_k$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\alpha_k = \\frac{\\boldsymbol{p}_k^T \\boldsymbol{b}}{\\boldsymbol{p}_k^T \\boldsymbol{A} \\boldsymbol{p}_k}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method and iterations\n", + "\n", + "If we choose the conjugate vectors $\\boldsymbol{p}_k$ carefully, \n", + "then we may not need all of them to obtain a good approximation to the solution \n", + "$\\boldsymbol{x}$. \n", + "We want to regard the conjugate gradient method as an iterative method. \n", + "This will us to solve systems where $n$ is so large that the direct \n", + "method would take too much time.\n", + "\n", + "We denote the initial guess for $\\boldsymbol{x}$ as $\\boldsymbol{x}_0$. \n", + "We can assume without loss of generality that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_0=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or consider the system" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{z} = \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "instead.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "One can show that the solution $\\boldsymbol{x}$ is also the unique minimizer of the quadratic form" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(\\boldsymbol{x}) = \\frac{1}{2}\\boldsymbol{x}^T\\boldsymbol{A}\\boldsymbol{x} - \\boldsymbol{x}^T \\boldsymbol{x} , \\quad \\boldsymbol{x}\\in\\mathbf{R}^n.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This suggests taking the first basis vector $\\boldsymbol{p}_1$ \n", + "to be the gradient of $f$ at $\\boldsymbol{x}=\\boldsymbol{x}_0$, \n", + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{x}_0-\\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and \n", + "$\\boldsymbol{x}_0=0$ it is equal $-\\boldsymbol{b}$.\n", + "The other vectors in the basis will be conjugate to the gradient, \n", + "hence the name conjugate gradient method.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "Let $\\boldsymbol{r}_k$ be the residual at the $k$-th step:" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_k=\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_k.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Note that $\\boldsymbol{r}_k$ is the negative gradient of $f$ at \n", + "$\\boldsymbol{x}=\\boldsymbol{x}_k$, \n", + "so the gradient descent method would be to move in the direction $\\boldsymbol{r}_k$. \n", + "Here, we insist that the directions $\\boldsymbol{p}_k$ are conjugate to each other, \n", + "so we take the direction closest to the gradient $\\boldsymbol{r}_k$ \n", + "under the conjugacy constraint. \n", + "This gives the following expression" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{p}_{k+1}=\\boldsymbol{r}_k-\\frac{\\boldsymbol{p}_k^T \\boldsymbol{A}\\boldsymbol{r}_k}{\\boldsymbol{p}_k^T\\boldsymbol{A}\\boldsymbol{p}_k} \\boldsymbol{p}_k.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method\n", + "We can also compute the residual iteratively as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_{k+1}=\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_{k+1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{b}-\\boldsymbol{A}(\\boldsymbol{x}_k+\\alpha_k\\boldsymbol{p}_k),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_k)-\\alpha_k\\boldsymbol{A}\\boldsymbol{p}_k,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which gives" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_{k+1}=\\boldsymbol{r}_k-\\boldsymbol{A}\\boldsymbol{p}_{k},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Revisiting our first homework\n", + "\n", + "We will use linear regression as a case study for the gradient descent\n", + "methods. Linear regression is a great test case for the gradient\n", + "descent methods discussed in the lectures since it has several\n", + "desirable properties such as:\n", + "\n", + "1. An analytical solution (recall homework set 1).\n", + "\n", + "2. The gradient can be computed analytically.\n", + "\n", + "3. The cost function is convex which guarantees that gradient descent converges for small enough learning rates\n", + "\n", + "We revisit an example similar to what we had in the first homework set. We had a function of the type" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "x = 2*np.random.rand(m,1)\n", + "y = 4+3*x+np.random.randn(m,1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $x_i \\in [0,1] $ is chosen randomly using a uniform distribution. Additionally we have a stochastic noise chosen according to a normal distribution $\\cal {N}(0,1)$. \n", + "The linear regression model is given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "h_\\beta(x) = \\boldsymbol{y} = \\beta_0 + \\beta_1 x,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "such that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y}_i = \\beta_0 + \\beta_1 x_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Gradient descent example\n", + "\n", + "Let $\\mathbf{y} = (y_1,\\cdots,y_n)^T$, $\\mathbf{\\boldsymbol{y}} = (\\boldsymbol{y}_1,\\cdots,\\boldsymbol{y}_n)^T$ and $\\beta = (\\beta_0, \\beta_1)^T$\n", + "\n", + "It is convenient to write $\\mathbf{\\boldsymbol{y}} = X\\beta$ where $X \\in \\mathbb{R}^{100 \\times 2} $ is the design matrix given by (we keep the intercept here)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "X \\equiv \\begin{bmatrix}\n", + "1 & x_1 \\\\\n", + "\\vdots & \\vdots \\\\\n", + "1 & x_{100} & \\\\\n", + "\\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The cost/loss/risk function is given by (" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\beta) = \\frac{1}{n}||X\\beta-\\mathbf{y}||_{2}^{2} = \\frac{1}{n}\\sum_{i=1}^{100}\\left[ (\\beta_0 + \\beta_1 x_i)^2 - 2 y_i (\\beta_0 + \\beta_1 x_i) + y_i^2\\right]\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and we want to find $\\beta$ such that $C(\\beta)$ is minimized.\n", + "\n", + "\n", + "## The derivative of the cost/loss function\n", + "\n", + "Computing $\\partial C(\\beta) / \\partial \\beta_0$ and $\\partial C(\\beta) / \\partial \\beta_1$ we can show that the gradient can be written as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_{\\beta} C(\\beta) = \\frac{2}{n}\\begin{bmatrix} \\sum_{i=1}^{100} \\left(\\beta_0+\\beta_1x_i-y_i\\right) \\\\\n", + "\\sum_{i=1}^{100}\\left( x_i (\\beta_0+\\beta_1x_i)-y_ix_i\\right) \\\\\n", + "\\end{bmatrix} = \\frac{2}{n}X^T(X\\beta - \\mathbf{y}),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $X$ is the design matrix defined above.\n", + "\n", + "\n", + "## The Hessian matrix\n", + "The Hessian matrix of $C(\\beta)$ is given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{H} \\equiv \\begin{bmatrix}\n", + "\\frac{\\partial^2 C(\\beta)}{\\partial \\beta_0^2} & \\frac{\\partial^2 C(\\beta)}{\\partial \\beta_0 \\partial \\beta_1} \\\\\n", + "\\frac{\\partial^2 C(\\beta)}{\\partial \\beta_0 \\partial \\beta_1} & \\frac{\\partial^2 C(\\beta)}{\\partial \\beta_1^2} & \\\\\n", + "\\end{bmatrix} = \\frac{2}{n}X^T X.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This result implies that $C(\\beta)$ is a convex function since the matrix $X^T X$ always is positive semi-definite.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Simple program\n", + "\n", + "We can now write a program that minimizes $C(\\beta)$ using the gradient descent method with a constant learning rate $\\gamma$ according to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_{k+1} = \\beta_k - \\gamma \\nabla_\\beta C(\\beta_k), \\ k=0,1,\\cdots\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can use the expression we computed for the gradient and let use a\n", + "$\\beta_0$ be chosen randomly and let $\\gamma = 0.001$. Stop iterating\n", + "when $||\\nabla_\\beta C(\\beta_k) || \\leq \\epsilon = 10^{-8}$. **Note that the code below does not include the latter stop criterion**.\n", + "\n", + "And finally we can compare our solution for $\\beta$ with the analytic result given by \n", + "$\\beta= (X^TX)^{-1} X^T \\mathbf{y}$.\n", + "\n", + "\n", + "## Gradient Descent Example\n", + "\n", + "Here our simple example" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\n", + "# Importing various packages\n", + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from mpl_toolkits.mplot3d import Axes3D\n", + "from matplotlib import cm\n", + "from matplotlib.ticker import LinearLocator, FormatStrFormatter\n", + "import sys\n", + "\n", + "# the number of datapoints\n", + "n = 100\n", + "x = 2*np.random.rand(n,1)\n", + "y = 4+3*x+np.random.randn(n,1)\n", + "\n", + "X = np.c_[np.ones((n,1)), x]\n", + "# Hessian matrix\n", + "H = (2.0/n)* X.T @ X\n", + "# Get the eigenvalues\n", + "EigValues, EigVectors = np.linalg.eig(H)\n", + "print(EigValues)\n", + "\n", + "beta_linreg = np.linalg.inv(X.T @ X) @ X.T @ y\n", + "print(beta_linreg)\n", + "beta = np.random.randn(2,1)\n", + "\n", + "eta = 1.0/np.max(EigValues)\n", + "Niterations = 1000\n", + "\n", + "for iter in range(Niterations):\n", + " gradient = (2.0/n)*X.T @ (X @ beta-y)\n", + " beta -= eta*gradient\n", + "\n", + "print(beta)\n", + "xnew = np.array([[0],[2]])\n", + "xbnew = np.c_[np.ones((2,1)), xnew]\n", + "ypredict = xbnew.dot(beta)\n", + "ypredict2 = xbnew.dot(beta_linreg)\n", + "plt.plot(xnew, ypredict, \"r-\")\n", + "plt.plot(xnew, ypredict2, \"b-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,2.0,0, 15.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Gradient descent example')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## And a corresponding example using **scikit-learn**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import SGDRegressor\n", + "\n", + "n = 100\n", + "x = 2*np.random.rand(n,1)\n", + "y = 4+3*x+np.random.randn(n,1)\n", + "\n", + "X = np.c_[np.ones((n,1)), x]\n", + "beta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)\n", + "print(beta_linreg)\n", + "sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1)\n", + "sgdreg.fit(x,y.ravel())\n", + "print(sgdreg.intercept_, sgdreg.coef_)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Gradient descent and Ridge\n", + "\n", + "We have also discussed Ridge regression where the loss function contains a regularized term given by the $L_2$ norm of $\\beta$," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C_{\\text{ridge}}(\\beta) = \\frac{1}{n}||X\\beta -\\mathbf{y}||^2 + \\lambda ||\\beta||^2, \\ \\lambda \\geq 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In order to minimize $C_{\\text{ridge}}(\\beta)$ using GD we only have adjust the gradient as follows" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_\\beta C_{\\text{ridge}}(\\beta) = \\frac{2}{n}\\begin{bmatrix} \\sum_{i=1}^{100} \\left(\\beta_0+\\beta_1x_i-y_i\\right) \\\\\n", + "\\sum_{i=1}^{100}\\left( x_i (\\beta_0+\\beta_1x_i)-y_ix_i\\right) \\\\\n", + "\\end{bmatrix} + 2\\lambda\\begin{bmatrix} \\beta_0 \\\\ \\beta_1\\end{bmatrix} = 2 (X^T(X\\beta - \\mathbf{y})+\\lambda \\beta).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can easily extend our program to minimize $C_{\\text{ridge}}(\\beta)$ using gradient descent and compare with the analytical solution given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_{\\text{ridge}} = \\left(X^T X + \\lambda I_{2 \\times 2} \\right)^{-1} X^T \\mathbf{y}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Program example for gradient descent with Ridge Regression" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from mpl_toolkits.mplot3d import Axes3D\n", + "from matplotlib import cm\n", + "from matplotlib.ticker import LinearLocator, FormatStrFormatter\n", + "import sys\n", + "\n", + "# the number of datapoints\n", + "n = 100\n", + "x = 2*np.random.rand(n,1)\n", + "y = 4+3*x+np.random.randn(n,1)\n", + "\n", + "X = np.c_[np.ones((n,1)), x]\n", + "XT_X = X.T @ X\n", + "\n", + "#Ridge parameter lambda\n", + "lmbda = 0.001\n", + "Id = lmbda* np.eye(XT_X.shape[0])\n", + "\n", + "beta_linreg = np.linalg.inv(XT_X+Id) @ X.T @ y\n", + "print(beta_linreg)\n", + "# Start plain gradient descent\n", + "beta = np.random.randn(2,1)\n", + "\n", + "eta = 0.1\n", + "Niterations = 100\n", + "\n", + "for iter in range(Niterations):\n", + " gradients = 2.0/n*X.T @ (X @ (beta)-y)+2*lmbda*beta\n", + " beta -= eta*gradients\n", + "\n", + "print(beta)\n", + "ypredict = X @ beta\n", + "ypredict2 = X @ beta_linreg\n", + "plt.plot(x, ypredict, \"r-\")\n", + "plt.plot(x, ypredict2, \"b-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,2.0,0, 15.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Gradient descent example for Ridge')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Using gradient descent methods, limitations\n", + "\n", + "* **Gradient descent (GD) finds local minima of our function**. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our cost/loss/risk function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.\n", + "\n", + "* **GD is sensitive to initial conditions**. One consequence of the local nature of GD is that initial conditions matter. Depending on where one starts, one will end up at a different local minima. Therefore, it is very important to think about how one initializes the training process. This is true for GD as well as more complicated variants of GD.\n", + "\n", + "* **Gradients are computationally expensive to calculate for large datasets**. In many cases in statistics and ML, the cost/loss/risk function is a sum of terms, with one term for each data point. For example, in linear regression, $E \\propto \\sum_{i=1}^n (y_i - \\mathbf{w}^T\\cdot\\mathbf{x}_i)^2$; for logistic regression, the square error is replaced by the cross entropy. To calculate the gradient we have to sum over *all* $n$ data points. Doing this at every GD step becomes extremely computationally expensive. An ingenious solution to this, is to calculate the gradients using small subsets of the data called \"mini batches\". This has the added benefit of introducing stochasticity into our algorithm.\n", + "\n", + "* **GD is very sensitive to choices of learning rates**. GD is extremely sensitive to the choice of learning rates. If the learning rate is very small, the training process take an extremely long time. For larger learning rates, GD can diverge and give poor results. Furthermore, depending on what the local landscape looks like, we have to modify the learning rates to ensure convergence. Ideally, we would *adaptively* choose the learning rates to match the landscape.\n", + "\n", + "* **GD treats all directions in parameter space uniformly.** Another major drawback of GD is that unlike Newton's method, the learning rate for GD is the same in all directions in parameter space. For this reason, the maximum learning rate is set by the behavior of the steepest direction and this can significantly slow down training. Ideally, we would like to take large steps in flat directions and small steps in steep directions. Since we are exploring rugged landscapes where curvatures change, this requires us to keep track of not only the gradient but second derivatives. The ideal scenario would be to calculate the Hessian but this proves to be too computationally expensive. \n", + "\n", + "* GD can take exponential time to escape saddle points, even with random initialization. As we mentioned, GD is extremely sensitive to initial condition since it determines the particular local minimum GD would eventually reach. However, even with a good initialization scheme, through the introduction of randomness, GD can still take exponential time to escape saddle points.\n", + "\n", + "## Stochastic Gradient Descent\n", + "\n", + "Stochastic gradient descent (SGD) and variants thereof address some of\n", + "the shortcomings of the Gradient descent method discussed above.\n", + "\n", + "The underlying idea of SGD comes from the observation that the cost\n", + "function, which we want to minimize, can almost always be written as a\n", + "sum over $n$ data points $\\{\\mathbf{x}_i\\}_{i=1}^n$," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\mathbf{\\beta}) = \\sum_{i=1}^n c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Computation of gradients\n", + "\n", + "This in turn means that the gradient can be\n", + "computed as a sum over $i$-gradients" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_\\beta C(\\mathbf{\\beta}) = \\sum_i^n \\nabla_\\beta c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Stochasticity/randomness is introduced by only taking the\n", + "gradient on a subset of the data called minibatches. If there are $n$\n", + "data points and the size of each minibatch is $M$, there will be $n/M$\n", + "minibatches. We denote these minibatches by $B_k$ where\n", + "$k=1,\\cdots,n/M$.\n", + "\n", + "\n", + "## SGD example\n", + "As an example, suppose we have $10$ data points $(\\mathbf{x}_1,\\cdots, \\mathbf{x}_{10})$ \n", + "and we choose to have $M=5$ minibathces,\n", + "then each minibatch contains two data points. In particular we have\n", + "$B_1 = (\\mathbf{x}_1,\\mathbf{x}_2), \\cdots, B_5 =\n", + "(\\mathbf{x}_9,\\mathbf{x}_{10})$. Note that if you choose $M=1$ you\n", + "have only a single batch with all data points and on the other extreme,\n", + "you may choose $M=n$ resulting in a minibatch for each datapoint, i.e\n", + "$B_k = \\mathbf{x}_k$.\n", + "\n", + "The idea is now to approximate the gradient by replacing the sum over\n", + "all data points with a sum over the data points in one the minibatches\n", + "picked at random in each gradient descent step" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_{\\beta}\n", + "C(\\mathbf{\\beta}) = \\sum_{i=1}^n \\nabla_\\beta c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta}) \\rightarrow \\sum_{i \\in B_k}^n \\nabla_\\beta\n", + "c_i(\\mathbf{x}_i, \\mathbf{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## The gradient step\n", + "\n", + "Thus a gradient descent step now looks like" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_{j+1} = \\beta_j - \\gamma_j \\sum_{i \\in B_k}^n \\nabla_\\beta c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta})\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $k$ is picked at random with equal\n", + "probability from $[1,n/M]$. An iteration over the number of\n", + "minibathces (n/M) is commonly referred to as an epoch. Thus it is\n", + "typical to choose a number of epochs and for each epoch iterate over\n", + "the number of minibatches, as exemplified in the code below.\n", + "\n", + "\n", + "## Simple example code" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np \n", + "\n", + "n = 100 #100 datapoints \n", + "M = 5 #size of each minibatch\n", + "m = int(n/M) #number of minibatches\n", + "n_epochs = 10 #number of epochs\n", + "\n", + "j = 0\n", + "for epoch in range(1,n_epochs+1):\n", + " for i in range(m):\n", + " k = np.random.randint(m) #Pick the k-th minibatch at random\n", + " #Compute the gradient using the data in minibatch Bk\n", + " #Compute new suggestion for \n", + " j += 1" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Taking the gradient only on a subset of the data has two important\n", + "benefits. First, it introduces randomness which decreases the chance\n", + "that our opmization scheme gets stuck in a local minima. Second, if\n", + "the size of the minibatches are small relative to the number of\n", + "datapoints ($M < n$), the computation of the gradient is much\n", + "cheaper since we sum over the datapoints in the $k-th$ minibatch and not\n", + "all $n$ datapoints.\n", + "\n", + "\n", + "## When do we stop?\n", + "\n", + "A natural question is when do we stop the search for a new minimum?\n", + "One possibility is to compute the full gradient after a given number\n", + "of epochs and check if the norm of the gradient is smaller than some\n", + "threshold and stop if true. However, the condition that the gradient\n", + "is zero is valid also for local minima, so this would only tell us\n", + "that we are close to a local/global minimum. However, we could also\n", + "evaluate the cost function at this point, store the result and\n", + "continue the search. If the test kicks in at a later stage we can\n", + "compare the values of the cost function and keep the $\\beta$ that\n", + "gave the lowest value.\n", + "\n", + "\n", + "## Slightly different approach\n", + "\n", + "Another approach is to let the step length $\\gamma_j$ depend on the\n", + "number of epochs in such a way that it becomes very small after a\n", + "reasonable time such that we do not move at all.\n", + "\n", + "As an example, let $e = 0,1,2,3,\\cdots$ denote the current epoch and let $t_0, t_1 > 0$ be two fixed numbers. Furthermore, let $t = e \\cdot m + i$ where $m$ is the number of minibatches and $i=0,\\cdots,m-1$. Then the function $$\\gamma_j(t; t_0, t_1) = \\frac{t_0}{t+t_1} $$ goes to zero as the number of epochs gets large. I.e. we start with a step length $\\gamma_j (0; t_0, t_1) = t_0/t_1$ which decays in *time* $t$.\n", + "\n", + "In this way we can fix the number of epochs, compute $\\beta$ and\n", + "evaluate the cost function at the end. Repeating the computation will\n", + "give a different result since the scheme is random by design. Then we\n", + "pick the final $\\beta$ that gives the lowest value of the cost\n", + "function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np \n", + "\n", + "def step_length(t,t0,t1):\n", + " return t0/(t+t1)\n", + "\n", + "n = 100 #100 datapoints \n", + "M = 5 #size of each minibatch\n", + "m = int(n/M) #number of minibatches\n", + "n_epochs = 500 #number of epochs\n", + "t0 = 1.0\n", + "t1 = 10\n", + "\n", + "gamma_j = t0/t1\n", + "j = 0\n", + "for epoch in range(1,n_epochs+1):\n", + " for i in range(m):\n", + " k = np.random.randint(m) #Pick the k-th minibatch at random\n", + " #Compute the gradient using the data in minibatch Bk\n", + " #Compute new suggestion for beta\n", + " t = epoch*m+i\n", + " gamma_j = step_length(t,t0,t1)\n", + " j += 1\n", + "\n", + "print(\"gamma_j after %d epochs: %g\" % (n_epochs,gamma_j))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Program for stochastic gradient" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "from math import exp, sqrt\n", + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import SGDRegressor\n", + "\n", + "m = 100\n", + "x = 2*np.random.rand(m,1)\n", + "y = 4+3*x+np.random.randn(m,1)\n", + "\n", + "X = np.c_[np.ones((m,1)), x]\n", + "theta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)\n", + "print(\"Own inversion\")\n", + "print(theta_linreg)\n", + "sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1)\n", + "sgdreg.fit(x,y.ravel())\n", + "print(\"sgdreg from scikit\")\n", + "print(sgdreg.intercept_, sgdreg.coef_)\n", + "\n", + "\n", + "theta = np.random.randn(2,1)\n", + "eta = 0.1\n", + "Niterations = 1000\n", + "\n", + "\n", + "for iter in range(Niterations):\n", + " gradients = 2.0/m*X.T @ ((X @ theta)-y)\n", + " theta -= eta*gradients\n", + "print(\"theta from own gd\")\n", + "print(theta)\n", + "\n", + "xnew = np.array([[0],[2]])\n", + "Xnew = np.c_[np.ones((2,1)), xnew]\n", + "ypredict = Xnew.dot(theta)\n", + "ypredict2 = Xnew.dot(theta_linreg)\n", + "\n", + "\n", + "n_epochs = 50\n", + "t0, t1 = 5, 50\n", + "def learning_schedule(t):\n", + " return t0/(t+t1)\n", + "\n", + "theta = np.random.randn(2,1)\n", + "\n", + "for epoch in range(n_epochs):\n", + " for i in range(m):\n", + " random_index = np.random.randint(m)\n", + " xi = X[random_index:random_index+1]\n", + " yi = y[random_index:random_index+1]\n", + " gradients = 2 * xi.T @ ((xi @ theta)-yi)\n", + " eta = learning_schedule(epoch*m+i)\n", + " theta = theta - eta*gradients\n", + "print(\"theta from own sdg\")\n", + "print(theta)\n", + "\n", + "plt.plot(xnew, ypredict, \"r-\")\n", + "plt.plot(xnew, ypredict2, \"b-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,2.0,0, 15.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Random numbers ')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "**Challenge**: try to write a similar code for a Logistic Regression case." + ] + } + ], + "metadata": { + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.5" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} \ No newline at end of file diff --git a/doc/LectureNotes/_build/jupyter_execute/chapter4.py b/doc/LectureNotes/_build/jupyter_execute/chapter4.py new file mode 100644 index 000000000..a01b04c67 --- /dev/null +++ b/doc/LectureNotes/_build/jupyter_execute/chapter4.py @@ -0,0 +1,1752 @@ +# Logistic Regression + + +[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK3155/h20/forelesningsvideoer/LectureSeptember18.mp4?vrtx=view-as-webpage) + + +## Logistic Regression + +In linear regression our main interest was centered on learning the +coefficients of a functional fit (say a polynomial) in order to be +able to predict the response of a continuous variable on some unseen +data. The fit to the continuous variable $y_i$ is based on some +independent variables $\hat{x}_i$. Linear regression resulted in +analytical expressions for standard ordinary Least Squares or Ridge +regression (in terms of matrices to invert) for several quantities, +ranging from the variance and thereby the confidence intervals of the +parameters $\hat{\beta}$ to the mean squared error. If we can invert +the product of the design matrices, linear regression gives then a +simple recipe for fitting our data. + + +Classification problems, however, are concerned with outcomes taking +the form of discrete variables (i.e. categories). We may for example, +on the basis of DNA sequencing for a number of patients, like to find +out which mutations are important for a certain disease; or based on +scans of various patients' brains, figure out if there is a tumor or +not; or given a specific physical system, we'd like to identify its +state, say whether it is an ordered or disordered system (typical +situation in solid state physics); or classify the status of a +patient, whether she/he has a stroke or not and many other similar +situations. + +The most common situation we encounter when we apply logistic +regression is that of two possible outcomes, normally denoted as a +binary outcome, true or false, positive or negative, success or +failure etc. + + +Logistic regression will also serve as our stepping stone towards +neural network algorithms and supervised deep learning. For logistic +learning, the minimization of the cost function leads to a non-linear +equation in the parameters $\hat{\beta}$. The optimization of the +problem calls therefore for minimization algorithms. This forms the +bottle neck of all machine learning algorithms, namely how to find +reliable minima of a multi-variable function. This leads us to the +family of gradient descent methods. The latter are the working horses +of basically all modern machine learning algorithms. + +We note also that many of the topics discussed here on logistic +regression are also commonly used in modern supervised Deep Learning +models, as we will see later. + + + +## Basics + +We consider the case where the dependent variables, also called the +responses or the outcomes, $y_i$ are discrete and only take values +from $k=0,\dots,K-1$ (i.e. $K$ classes). + +The goal is to predict the +output classes from the design matrix $\hat{X}\in\mathbb{R}^{n\times p}$ +made of $n$ samples, each of which carries $p$ features or predictors. The +primary goal is to identify the classes to which new unseen samples +belong. + +Let us specialize to the case of two classes only, with outputs +$y_i=0$ and $y_i=1$. Our outcomes could represent the status of a +credit card user that could default or not on her/his credit card +debt. That is + +$$ +y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}. +$$ + +Before moving to the logistic model, let us try to use our linear +regression model to classify these two outcomes. We could for example +fit a linear model to the default case if $y_i > 0.5$ and the no +default case $y_i \leq 0.5$. + +We would then have our +weighted linear combination, namely + + +
+ +$$ +\begin{equation} +\hat{y} = \hat{X}^T\hat{\beta} + \hat{\epsilon}, +\label{_auto1} \tag{1} +\end{equation} +$$ + +where $\hat{y}$ is a vector representing the possible outcomes, $\hat{X}$ is our +$n\times p$ design matrix and $\hat{\beta}$ represents our estimators/predictors. + + +The main problem with our function is that it takes values on the +entire real axis. In the case of logistic regression, however, the +labels $y_i$ are discrete variables. A typical example is the credit +card data discussed below here, where we can set the state of +defaulting the debt to $y_i=1$ and not to $y_i=0$ for one the persons +in the data set (see the full example below). + +One simple way to get a discrete output is to have sign +functions that map the output of a linear regressor to values $\{0,1\}$, +$f(s_i)=sign(s_i)=1$ if $s_i\ge 0$ and 0 if otherwise. +We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron" model in the machine learning +literature. This model is extremely simple. However, in many cases it is more +favorable to use a ``soft" classifier that outputs +the probability of a given category. This leads us to the logistic function. + + +The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful. + +%matplotlib inline + +# Common imports +import os +import numpy as np +import pandas as pd +import matplotlib.pyplot as plt +from sklearn.linear_model import LinearRegression, Ridge, Lasso +from sklearn.model_selection import train_test_split +from sklearn.utils import resample +from sklearn.metrics import mean_squared_error +from IPython.display import display +from pylab import plt, mpl +plt.style.use('seaborn') +mpl.rcParams['font.family'] = 'serif' + +# Where to save the figures and data files +PROJECT_ROOT_DIR = "Results" +FIGURE_ID = "Results/FigureFiles" +DATA_ID = "DataFiles/" + +if not os.path.exists(PROJECT_ROOT_DIR): + os.mkdir(PROJECT_ROOT_DIR) + +if not os.path.exists(FIGURE_ID): + os.makedirs(FIGURE_ID) + +if not os.path.exists(DATA_ID): + os.makedirs(DATA_ID) + +def image_path(fig_id): + return os.path.join(FIGURE_ID, fig_id) + +def data_path(dat_id): + return os.path.join(DATA_ID, dat_id) + +def save_fig(fig_id): + plt.savefig(image_path(fig_id) + ".png", format='png') + +infile = open(data_path("chddata.csv"),'r') + +# Read the chd data as csv file and organize the data into arrays with age group, age, and chd +chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD')) +chd.columns = ['ID', 'Age', 'Agegroup', 'CHD'] +output = chd['CHD'] +age = chd['Age'] +agegroup = chd['Agegroup'] +numberID = chd['ID'] +display(chd) + +plt.scatter(age, output, marker='o') +plt.axis([18,70.0,-0.1, 1.2]) +plt.xlabel(r'Age') +plt.ylabel(r'CHD') +plt.title(r'Age distribution and Coronary heart disease') +plt.show() + +What we could attempt however is to plot the mean value for each group. + +agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800]) +group = np.array([1, 2, 3, 4, 5, 6, 7, 8]) +plt.plot(group, agegroupmean, "r-") +plt.axis([0,9,0, 1.0]) +plt.xlabel(r'Age group') +plt.ylabel(r'CHD mean values') +plt.title(r'Mean values for each age group') +plt.show() + +We are now trying to find a function $f(y\vert x)$, that is a function which gives us an expected value for the output $y$ with a given input $x$. +In standard linear regression with a linear dependence on $x$, we would write this in terms of our model + +$$ +f(y_i\vert x_i)=\beta_0+\beta_1 x_i. +$$ + +This expression implies however that $f(y_i\vert x_i)$ could take any +value from minus infinity to plus infinity. If we however let +$f(y\vert y)$ be represented by the mean value, the above example +shows us that we can constrain the function to take values between +zero and one, that is we have $0 \le f(y_i\vert x_i) \le 1$. Looking +at our last curve we see also that it has an S-shaped form. This leads +us to a very popular model for the function $f$, namely the so-called +Sigmoid function or logistic model. We will consider this function as +representing the probability for finding a value of $y_i$ with a given +$x_i$. + + +## The logistic function + +Another widely studied model, is the so-called +perceptron model, which is an example of a "hard classification" model. We +will encounter this model when we discuss neural networks as +well. Each datapoint is deterministically assigned to a category (i.e +$y_i=0$ or $y_i=1$). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a "soft" +classifier that outputs the probability of a given category rather +than a single value. For example, given $x_i$, the classifier +outputs the probability of being in a category $k$. Logistic regression +is the most common example of a so-called soft classifier. In logistic +regression, the probability that a data point $x_i$ +belongs to a category $y_i=\{0,1\}$ is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event, + +$$ +p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}. +$$ + +Note that $1-p(t)= p(-t)$. + +## Examples of likelihood functions used in logistic regression and nueral networks + + +The following code plots the logistic function, the step function and other functions we will encounter from here and on. + +"""The sigmoid function (or the logistic curve) is a +function that takes any real number, z, and outputs a number (0,1). +It is useful in neural networks for assigning weights on a relative scale. +The value z is the weighted sum of parameters involved in the learning algorithm.""" + +import numpy +import matplotlib.pyplot as plt +import math as mt + +z = numpy.arange(-5, 5, .1) +sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z))) +sigma = sigma_fn(z) + +fig = plt.figure() +ax = fig.add_subplot(111) +ax.plot(z, sigma) +ax.set_ylim([-0.1, 1.1]) +ax.set_xlim([-5,5]) +ax.grid(True) +ax.set_xlabel('z') +ax.set_title('sigmoid function') + +plt.show() + +"""Step Function""" +z = numpy.arange(-5, 5, .02) +step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0) +step = step_fn(z) + +fig = plt.figure() +ax = fig.add_subplot(111) +ax.plot(z, step) +ax.set_ylim([-0.5, 1.5]) +ax.set_xlim([-5,5]) +ax.grid(True) +ax.set_xlabel('z') +ax.set_title('step function') + +plt.show() + +"""tanh Function""" +z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1) +t = numpy.tanh(z) + +fig = plt.figure() +ax = fig.add_subplot(111) +ax.plot(z, t) +ax.set_ylim([-1.0, 1.0]) +ax.set_xlim([-2*mt.pi,2*mt.pi]) +ax.grid(True) +ax.set_xlabel('z') +ax.set_title('tanh function') + +plt.show() + +We assume now that we have two classes with $y_i$ either $0$ or $1$. Furthermore we assume also that we have only two parameters $\beta$ in our fitting of the Sigmoid function, that is we define probabilities + +$$ +\begin{align*} +p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\ +p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}), +\end{align*} +$$ + +where $\hat{\beta}$ are the weights we wish to extract from data, in our case $\beta_0$ and $\beta_1$. + +Note that we used + +$$ +p(y_i=0\vert x_i, \hat{\beta}) = 1-p(y_i=1\vert x_i, \hat{\beta}). +$$ + +In order to define the total likelihood for all possible outcomes from a +dataset $\mathcal{D}=\{(y_i,x_i)\}$, with the binary labels +$y_i\in\{0,1\}$ and where the data points are drawn independently, we use the so-called [Maximum Likelihood Estimation](https://en.wikipedia.org/wiki/Maximum_likelihood_estimation) (MLE) principle. +We aim thus at maximizing +the probability of seeing the observed data. We can then approximate the +likelihood in terms of the product of the individual probabilities of a specific outcome $y_i$, that is + +$$ +\begin{align*} +P(\mathcal{D}|\hat{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\hat{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\hat{\beta}))\right]^{1-y_i}\nonumber \\ +\end{align*} +$$ + +from which we obtain the log-likelihood and our **cost/loss** function + +$$ +\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\hat{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\hat{\beta}))\right]\right). +$$ + +Reordering the logarithms, we can rewrite the **cost/loss** function as + +$$ +\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right). +$$ + +The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to $\beta$. +Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that + +$$ +\mathcal{C}(\hat{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right). +$$ + +This equation is known in statistics as the **cross entropy**. Finally, we note that just as in linear regression, +in practice we often supplement the cross-entropy with additional regularization terms, usually $L_1$ and $L_2$ regularization as we did for Ridge and Lasso regression. + + +The cross entropy is a convex function of the weights $\hat{\beta}$ and, +therefore, any local minimizer is a global minimizer. + + +Minimizing this +cost function with respect to the two parameters $\beta_0$ and $\beta_1$ we obtain + +$$ +\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right), +$$ + +and + +$$ +\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right). +$$ + +Let us now define a vector $\hat{y}$ with $n$ elements $y_i$, an +$n\times p$ matrix $\hat{X}$ which contains the $x_i$ values and a +vector $\hat{p}$ of fitted probabilities $p(y_i\vert x_i,\hat{\beta})$. We can rewrite in a more compact form the first +derivative of cost function as + +$$ +\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right). +$$ + +If we in addition define a diagonal matrix $\hat{W}$ with elements +$p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta})$, we can obtain a compact expression of the second derivative as + +$$ +\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}. +$$ + +Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with $p$ predictors + +$$ +\log{ \frac{p(\hat{\beta}\hat{x})}{1-p(\hat{\beta}\hat{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p. +$$ + +Here we defined $\hat{x}=[1,x_1,x_2,\dots,x_p]$ and $\hat{\beta}=[\beta_0, \beta_1, \dots, \beta_p]$ leading to + +$$ +p(\hat{\beta}\hat{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}. +$$ + +Till now we have mainly focused on two classes, the so-called binary +system. Suppose we wish to extend to $K$ classes. Let us for the sake +of simplicity assume we have only two predictors. We have then following model + +$$ +\log{\frac{p(C=1\vert x)}{p(K\vert x)}} = \beta_{10}+\beta_{11}x_1, +$$ + +and + +$$ +\log{\frac{p(C=2\vert x)}{p(K\vert x)}} = \beta_{20}+\beta_{21}x_1, +$$ + +and so on till the class $C=K-1$ class + +$$ +\log{\frac{p(C=K-1\vert x)}{p(K\vert x)}} = \beta_{(K-1)0}+\beta_{(K-1)1}x_1, +$$ + +and the model is specified in term of $K-1$ so-called log-odds or +**logit** transformations. + + + +In our discussion of neural networks we will encounter the above again +in terms of a slightly modified function, the so-called **Softmax** function. + +The softmax function is used in various multiclass classification +methods, such as multinomial logistic regression (also known as +softmax regression), multiclass linear discriminant analysis, naive +Bayes classifiers, and artificial neural networks. Specifically, in +multinomial logistic regression and linear discriminant analysis, the +input to the function is the result of $K$ distinct linear functions, +and the predicted probability for the $k$-th class given a sample +vector $\hat{x}$ and a weighting vector $\hat{\beta}$ is (with two +predictors): + +$$ +p(C=k\vert \mathbf {x} )=\frac{\exp{(\beta_{k0}+\beta_{k1}x_1)}}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}}. +$$ + +It is easy to extend to more predictors. The final class is + +$$ +p(C=K\vert \mathbf {x} )=\frac{1}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}}, +$$ + +and they sum to one. Our earlier discussions were all specialized to +the case with two classes only. It is easy to see from the above that +what we derived earlier is compatible with these equations. + +To find the optimal parameters we would typically use a gradient +descent method. Newton's method and gradient descent methods are +discussed in the material on [optimization +methods](https://compphysics.github.io/MachineLearning/doc/pub/Splines/html/Splines-bs.html). + +## Wisconsin Cancer Data + +We show here how we can use a simple regression case on the breast +cancer data using Logistic regression as our algorithm for +classification. + +import matplotlib.pyplot as plt +import numpy as np +from sklearn.model_selection import train_test_split +from sklearn.datasets import load_breast_cancer +from sklearn.linear_model import LogisticRegression + +# Load the data +cancer = load_breast_cancer() + +X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0) +print(X_train.shape) +print(X_test.shape) +# Logistic Regression +logreg = LogisticRegression(solver='lbfgs') +logreg.fit(X_train, y_train) +print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test))) +#now scale the data +from sklearn.preprocessing import StandardScaler +scaler = StandardScaler() +scaler.fit(X_train) +X_train_scaled = scaler.transform(X_train) +X_test_scaled = scaler.transform(X_test) +# Logistic Regression +logreg.fit(X_train_scaled, y_train) +print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test))) + +In addition to the above scores, we could also study the covariance (and the correlation matrix). +We use **Pandas** to compute the correlation matrix. + +import matplotlib.pyplot as plt +import numpy as np +from sklearn.model_selection import train_test_split +from sklearn.datasets import load_breast_cancer +from sklearn.linear_model import LogisticRegression +cancer = load_breast_cancer() +import pandas as pd +# Making a data frame +cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names) + +fig, axes = plt.subplots(15,2,figsize=(10,20)) +malignant = cancer.data[cancer.target == 0] +benign = cancer.data[cancer.target == 1] +ax = axes.ravel() + +for i in range(30): + _, bins = np.histogram(cancer.data[:,i], bins =50) + ax[i].hist(malignant[:,i], bins = bins, alpha = 0.5) + ax[i].hist(benign[:,i], bins = bins, alpha = 0.5) + ax[i].set_title(cancer.feature_names[i]) + ax[i].set_yticks(()) +ax[0].set_xlabel("Feature magnitude") +ax[0].set_ylabel("Frequency") +ax[0].legend(["Malignant", "Benign"], loc ="best") +fig.tight_layout() +plt.show() + +import seaborn as sns +correlation_matrix = cancerpd.corr().round(1) +# use the heatmap function from seaborn to plot the correlation matrix +# annot = True to print the values inside the square +plt.figure(figsize=(15,8)) +sns.heatmap(data=correlation_matrix, annot=True) +plt.show() + +In the above example we note two things. In the first plot we display +the overlap of benign and malignant tumors as functions of the various +features in the Wisconsing breast cancer data set. We see that for +some of the features we can distinguish clearly the benign and +malignant cases while for other features we cannot. This can point to +us which features may be of greater interest when we wish to classify +a benign or not benign tumour. + +In the second figure we have computed the so-called correlation +matrix, which in our case with thirty features becomes a $30\times 30$ +matrix. + +We constructed this matrix using **pandas** via the statements + +cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names) + +and then + +correlation_matrix = cancerpd.corr().round(1) + +Diagonalizing this matrix we can in turn say something about which +features are of relevance and which are not. This leads us to +the classical Principal Component Analysis (PCA) theorem with +applications. This will be discussed later this semester ([week 43](https://compphysics.github.io/MachineLearning/doc/pub/week43/html/week43-bs.html)). + +import matplotlib.pyplot as plt +import numpy as np +from sklearn.model_selection import train_test_split +from sklearn.datasets import load_breast_cancer +from sklearn.linear_model import LogisticRegression + +# Load the data +cancer = load_breast_cancer() + +X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0) +print(X_train.shape) +print(X_test.shape) +# Logistic Regression +logreg = LogisticRegression(solver='lbfgs') +logreg.fit(X_train, y_train) +print("Test set accuracy with Logistic Regression: {:.2f}".format(logreg.score(X_test,y_test))) +#now scale the data +from sklearn.preprocessing import StandardScaler +scaler = StandardScaler() +scaler.fit(X_train) +X_train_scaled = scaler.transform(X_train) +X_test_scaled = scaler.transform(X_test) +# Logistic Regression +logreg.fit(X_train_scaled, y_train) +print("Test set accuracy Logistic Regression with scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test))) + + +from sklearn.preprocessing import LabelEncoder +from sklearn.model_selection import cross_validate +#Cross validation +accuracy = cross_validate(logreg,X_test_scaled,y_test,cv=10)['test_score'] +print(accuracy) +print("Test set accuracy with Logistic Regression and scaled data: {:.2f}".format(logreg.score(X_test_scaled,y_test))) + + +import scikitplot as skplt +y_pred = logreg.predict(X_test_scaled) +skplt.metrics.plot_confusion_matrix(y_test, y_pred, normalize=True) +plt.show() +y_probas = logreg.predict_proba(X_test_scaled) +skplt.metrics.plot_roc(y_test, y_probas) +plt.show() +skplt.metrics.plot_cumulative_gain(y_test, y_probas) +plt.show() + +## Optimization, the central part of any Machine Learning algortithm + +Almost every problem in machine learning and data science starts with +a dataset $X$, a model $g(\beta)$, which is a function of the +parameters $\beta$ and a cost function $C(X, g(\beta))$ that allows +us to judge how well the model $g(\beta)$ explains the observations +$X$. The model is fit by finding the values of $\beta$ that minimize +the cost function. Ideally we would be able to solve for $\beta$ +analytically, however this is not possible in general and we must use +some approximative/numerical method to compute the minimum. + + + +## Revisiting our Logistic Regression case + +In our discussion on Logistic Regression we studied the +case of +two classes, with $y_i$ either +$0$ or $1$. Furthermore we assumed also that we have only two +parameters $\beta$ in our fitting, that is we +defined probabilities + +$$ +\begin{align*} +p(y_i=1|x_i,\boldsymbol{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\ +p(y_i=0|x_i,\boldsymbol{\beta}) &= 1 - p(y_i=1|x_i,\boldsymbol{\beta}), +\end{align*} +$$ + +where $\boldsymbol{\beta}$ are the weights we wish to extract from data, in our case $\beta_0$ and $\beta_1$. + + +## The equations to solve + +Our compact equations used a definition of a vector $\boldsymbol{y}$ with $n$ +elements $y_i$, an $n\times p$ matrix $\boldsymbol{X}$ which contains the +$x_i$ values and a vector $\boldsymbol{p}$ of fitted probabilities +$p(y_i\vert x_i,\boldsymbol{\beta})$. We rewrote in a more compact form +the first derivative of the cost function as + +$$ +\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}} = -\boldsymbol{X}^T\left(\boldsymbol{y}-\boldsymbol{p}\right). +$$ + +If we in addition define a diagonal matrix $\boldsymbol{W}$ with elements +$p(y_i\vert x_i,\boldsymbol{\beta})(1-p(y_i\vert x_i,\boldsymbol{\beta})$, we can obtain a compact expression of the second derivative as + +$$ +\frac{\partial^2 \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}\partial \boldsymbol{\beta}^T} = \boldsymbol{X}^T\boldsymbol{W}\boldsymbol{X}. +$$ + +This defines what is called the Hessian matrix. + + +## Solving using Newton-Raphson's method + +If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives. + +Our iterative scheme is then given by + +$$ +\boldsymbol{\beta}^{\mathrm{new}} = \boldsymbol{\beta}^{\mathrm{old}}-\left(\frac{\partial^2 \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}\partial \boldsymbol{\beta}^T}\right)^{-1}_{\boldsymbol{\beta}^{\mathrm{old}}}\times \left(\frac{\partial \mathcal{C}(\boldsymbol{\beta})}{\partial \boldsymbol{\beta}}\right)_{\boldsymbol{\beta}^{\mathrm{old}}}, +$$ + +or in matrix form as + +$$ +\boldsymbol{\beta}^{\mathrm{new}} = \boldsymbol{\beta}^{\mathrm{old}}-\left(\boldsymbol{X}^T\boldsymbol{W}\boldsymbol{X} \right)^{-1}\times \left(-\boldsymbol{X}^T(\boldsymbol{y}-\boldsymbol{p}) \right)_{\boldsymbol{\beta}^{\mathrm{old}}}. +$$ + +The right-hand side is computed with the old values of $\beta$. + +If we can compute these matrices, in particular the Hessian, the above is often the easiest method to implement. + + + +## Brief reminder on Newton-Raphson's method + +Let us quickly remind ourselves how we derive the above method. + +Perhaps the most celebrated of all one-dimensional root-finding +routines is Newton's method, also called the Newton-Raphson +method. This method requires the evaluation of both the +function $f$ and its derivative $f'$ at arbitrary points. +If you can only calculate the derivative +numerically and/or your function is not of the smooth type, we +normally discourage the use of this method. + + +## The equations + +The Newton-Raphson formula consists geometrically of extending the +tangent line at a current point until it crosses zero, then setting +the next guess to the abscissa of that zero-crossing. The mathematics +behind this method is rather simple. Employing a Taylor expansion for +$x$ sufficiently close to the solution $s$, we have + + +
+ +$$ +f(s)=0=f(x)+(s-x)f'(x)+\frac{(s-x)^2}{2}f''(x) +\dots. + \label{eq:taylornr} \tag{2} +$$ + +For small enough values of the function and for well-behaved +functions, the terms beyond linear are unimportant, hence we obtain + +$$ +f(x)+(s-x)f'(x)\approx 0, +$$ + +yielding + +$$ +s\approx x-\frac{f(x)}{f'(x)}. +$$ + +Having in mind an iterative procedure, it is natural to start iterating with + +$$ +x_{n+1}=x_n-\frac{f(x_n)}{f'(x_n)}. +$$ + +## Simple geometric interpretation + +The above is Newton-Raphson's method. It has a simple geometric +interpretation, namely $x_{n+1}$ is the point where the tangent from +$(x_n,f(x_n))$ crosses the $x$-axis. Close to the solution, +Newton-Raphson converges fast to the desired result. However, if we +are far from a root, where the higher-order terms in the series are +important, the Newton-Raphson formula can give grossly inaccurate +results. For instance, the initial guess for the root might be so far +from the true root as to let the search interval include a local +maximum or minimum of the function. If an iteration places a trial +guess near such a local extremum, so that the first derivative nearly +vanishes, then Newton-Raphson may fail totally + + + +## Extending to more than one variable + +Newton's method can be generalized to systems of several non-linear equations +and variables. Consider the case with two equations + +$$ +\begin{array}{cc} f_1(x_1,x_2) &=0\\ + f_2(x_1,x_2) &=0,\end{array} +$$ + +which we Taylor expand to obtain + +$$ +\begin{array}{cc} 0=f_1(x_1+h_1,x_2+h_2)=&f_1(x_1,x_2)+h_1 + \partial f_1/\partial x_1+h_2 + \partial f_1/\partial x_2+\dots\\ + 0=f_2(x_1+h_1,x_2+h_2)=&f_2(x_1,x_2)+h_1 + \partial f_2/\partial x_1+h_2 + \partial f_2/\partial x_2+\dots + \end{array}. +$$ + +Defining the Jacobian matrix ${\bf \boldsymbol{J}}$ we have + +$$ +{\bf \boldsymbol{J}}=\left( \begin{array}{cc} + \partial f_1/\partial x_1 & \partial f_1/\partial x_2 \\ + \partial f_2/\partial x_1 &\partial f_2/\partial x_2 + \end{array} \right), +$$ + +we can rephrase Newton's method as + +$$ +\left(\begin{array}{c} x_1^{n+1} \\ x_2^{n+1} \end{array} \right)= +\left(\begin{array}{c} x_1^{n} \\ x_2^{n} \end{array} \right)+ +\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right), +$$ + +where we have defined + +$$ +\left(\begin{array}{c} h_1^{n} \\ h_2^{n} \end{array} \right)= + -{\bf \boldsymbol{J}}^{-1} + \left(\begin{array}{c} f_1(x_1^{n},x_2^{n}) \\ f_2(x_1^{n},x_2^{n}) \end{array} \right). +$$ + +We need thus to compute the inverse of the Jacobian matrix and it +is to understand that difficulties may +arise in case ${\bf \boldsymbol{J}}$ is nearly singular. + +It is rather straightforward to extend the above scheme to systems of +more than two non-linear equations. In our case, the Jacobian matrix is given by the Hessian that represents the second derivative of cost function. + + + + +## Steepest descent + +The basic idea of gradient descent is +that a function $F(\mathbf{x})$, +$\mathbf{x} \equiv (x_1,\cdots,x_n)$, decreases fastest if one goes from $\bf {x}$ in the +direction of the negative gradient $-\nabla F(\mathbf{x})$. + +It can be shown that if + +$$ +\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), +$$ + +with $\gamma_k > 0$. + +For $\gamma_k$ small enough, then $F(\mathbf{x}_{k+1}) \leq +F(\mathbf{x}_k)$. This means that for a sufficiently small $\gamma_k$ +we are always moving towards smaller function values, i.e a minimum. + + +## More on Steepest descent + +The previous observation is the basis of the method of steepest +descent, which is also referred to as just gradient descent (GD). One +starts with an initial guess $\mathbf{x}_0$ for a minimum of $F$ and +computes new approximations according to + +$$ +\mathbf{x}_{k+1} = \mathbf{x}_k - \gamma_k \nabla F(\mathbf{x}_k), \ \ k \geq 0. +$$ + +The parameter $\gamma_k$ is often referred to as the step length or +the learning rate within the context of Machine Learning. + + +## The ideal + +Ideally the sequence $\{\mathbf{x}_k \}_{k=0}$ converges to a global +minimum of the function $F$. In general we do not know if we are in a +global or local minimum. In the special case when $F$ is a convex +function, all local minima are also global minima, so in this case +gradient descent can converge to the global solution. The advantage of +this scheme is that it is conceptually simple and straightforward to +implement. However the method in this form has some severe +limitations: + +In machine learing we are often faced with non-convex high dimensional +cost functions with many local minima. Since GD is deterministic we +will get stuck in a local minimum, if the method converges, unless we +have a very good intial guess. This also implies that the scheme is +sensitive to the chosen initial condition. + +Note that the gradient is a function of $\mathbf{x} = +(x_1,\cdots,x_n)$ which makes it expensive to compute numerically. + + + +## The sensitiveness of the gradient descent + +The gradient descent method +is sensitive to the choice of learning rate $\gamma_k$. This is due +to the fact that we are only guaranteed that $F(\mathbf{x}_{k+1}) \leq +F(\mathbf{x}_k)$ for sufficiently small $\gamma_k$. The problem is to +determine an optimal learning rate. If the learning rate is chosen too +small the method will take a long time to converge and if it is too +large we can experience erratic behavior. + +Many of these shortcomings can be alleviated by introducing +randomness. One such method is that of Stochastic Gradient Descent +(SGD), see below. + + + +## Convex functions + +Ideally we want our cost/loss function to be convex(concave). + +First we give the definition of a convex set: A set $C$ in +$\mathbb{R}^n$ is said to be convex if, for all $x$ and $y$ in $C$ and +all $t \in (0,1)$ , the point $(1 − t)x + ty$ also belongs to +C. Geometrically this means that every point on the line segment +connecting $x$ and $y$ is in $C$ as discussed below. + +The convex subsets of $\mathbb{R}$ are the intervals of +$\mathbb{R}$. Examples of convex sets of $\mathbb{R}^2$ are the +regular polygons (triangles, rectangles, pentagons, etc...). + + +## Convex function + +**Convex function**: Let $X \subset \mathbb{R}^n$ be a convex set. Assume that the function $f: X \rightarrow \mathbb{R}$ is continuous, then $f$ is said to be convex if $$f(tx_1 + (1-t)x_2) \leq tf(x_1) + (1-t)f(x_2) $$ for all $x_1, x_2 \in X$ and for all $t \in [0,1]$. If $\leq$ is replaced with a strict inequaltiy in the definition, we demand $x_1 \neq x_2$ and $t\in(0,1)$ then $f$ is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting $f(x_1)$ and $f(x_2)$, the value of the function on the interval $[x_1,x_2]$ is always below the line as illustrated below. + + +## Conditions on convex functions + +In the following we state first and second-order conditions which +ensures convexity of a function $f$. We write $D_f$ to denote the +domain of $f$, i.e the subset of $R^n$ where $f$ is defined. For more +details and proofs we refer to: [S. Boyd and L. Vandenberghe. Convex Optimization. Cambridge University Press](http://stanford.edu/boyd/cvxbook/, 2004). + +**First order condition.** + +Suppose $f$ is differentiable (i.e $\nabla f(x)$ is well defined for +all $x$ in the domain of $f$). Then $f$ is convex if and only if $D_f$ +is a convex set and $$f(y) \geq f(x) + \nabla f(x)^T (y-x) $$ holds +for all $x,y \in D_f$. This condition means that for a convex function +the first order Taylor expansion (right hand side above) at any point +a global under estimator of the function. To convince yourself you can +make a drawing of $f(x) = x^2+1$ and draw the tangent line to $f(x)$ and +note that it is always below the graph. + + + +**Second order condition.** + +Assume that $f$ is twice +differentiable, i.e the Hessian matrix exists at each point in +$D_f$. Then $f$ is convex if and only if $D_f$ is a convex set and its +Hessian is positive semi-definite for all $x\in D_f$. For a +single-variable function this reduces to $f''(x) \geq 0$. Geometrically this means that $f$ has nonnegative curvature +everywhere. + + + +This condition is particularly useful since it gives us an procedure for determining if the function under consideration is convex, apart from using the definition. + + +## More on convex functions + +The next result is of great importance to us and the reason why we are +going on about convex functions. In machine learning we frequently +have to minimize a loss/cost function in order to find the best +parameters for the model we are considering. + +Ideally we want the +global minimum (for high-dimensional models it is hard to know +if we have local or global minimum). However, if the cost/loss function +is convex the following result provides invaluable information: + +**Any minimum is global for convex functions.** + +Consider the problem of finding $x \in \mathbb{R}^n$ such that $f(x)$ +is minimal, where $f$ is convex and differentiable. Then, any point +$x^*$ that satisfies $\nabla f(x^*) = 0$ is a global minimum. + + + +This result means that if we know that the cost/loss function is convex and we are able to find a minimum, we are guaranteed that it is a global minimum. + + +## Some simple problems + +1. Show that $f(x)=x^2$ is convex for $x \in \mathbb{R}$ using the definition of convexity. Hint: If you re-write the definition, $f$ is convex if the following holds for all $x,y \in D_f$ and any $\lambda \in [0,1]$ $\lambda f(x)+(1-\lambda)f(y)-f(\lambda x + (1-\lambda) y ) \geq 0$. + +2. Using the second order condition show that the following functions are convex on the specified domain. + + * $f(x) = e^x$ is convex for $x \in \mathbb{R}$. + + * $g(x) = -\ln(x)$ is convex for $x \in (0,\infty)$. + + +3. Let $f(x) = x^2$ and $g(x) = e^x$. Show that $f(g(x))$ and $g(f(x))$ is convex for $x \in \mathbb{R}$. Also show that if $f(x)$ is any convex function than $h(x) = e^{f(x)}$ is convex. + +4. A norm is any function that satisfy the following properties + + * $f(\alpha x) = |\alpha| f(x)$ for all $\alpha \in \mathbb{R}$. + + * $f(x+y) \leq f(x) + f(y)$ + + * $f(x) \leq 0$ for all $x \in \mathbb{R}^n$ with equality if and only if $x = 0$ + + +Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this). + + + +## Friday September 25 + +[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember25.mp4?vrtx=view-as-webpage) and [link to handwritten notes](https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/NotesSeptember25.pdf). + + + +## Standard steepest descent + + +Before we proceed, we would like to discuss the approach called the +**standard Steepest descent** (different from the above steepest descent discussion), which again leads to us having to be able +to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG). + +[The success of the CG method](https://www.cs.cmu.edu/~quake-papers/painless-conjugate-gradient.pdf) +for finding solutions of non-linear problems is based on the theory +of conjugate gradients for linear systems of equations. It belongs to +the class of iterative methods for solving problems from linear +algebra of the type + +$$ +\boldsymbol{A}\boldsymbol{x} = \boldsymbol{b}. +$$ + +In the iterative process we end up with a problem like + +$$ +\boldsymbol{r}= \boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}, +$$ + +where $\boldsymbol{r}$ is the so-called residual or error in the iterative process. + +When we have found the exact solution, $\boldsymbol{r}=0$. + + +## Gradient method + +The residual is zero when we reach the minimum of the quadratic equation + +$$ +P(\boldsymbol{x})=\frac{1}{2}\boldsymbol{x}^T\boldsymbol{A}\boldsymbol{x} - \boldsymbol{x}^T\boldsymbol{b}, +$$ + +with the constraint that the matrix $\boldsymbol{A}$ is positive definite and +symmetric. This defines also the Hessian and we want it to be positive definite. + + + +## Steepest descent method + +We denote the initial guess for $\boldsymbol{x}$ as $\boldsymbol{x}_0$. +We can assume without loss of generality that + +$$ +\boldsymbol{x}_0=0, +$$ + +or consider the system + +$$ +\boldsymbol{A}\boldsymbol{z} = \boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_0, +$$ + +instead. + + + +## Steepest descent method +One can show that the solution $\boldsymbol{x}$ is also the unique minimizer of the quadratic form + +$$ +f(\boldsymbol{x}) = \frac{1}{2}\boldsymbol{x}^T\boldsymbol{A}\boldsymbol{x} - \boldsymbol{x}^T \boldsymbol{x} , \quad \boldsymbol{x}\in\mathbf{R}^n. +$$ + +This suggests taking the first basis vector $\boldsymbol{r}_1$ (see below for definition) +to be the gradient of $f$ at $\boldsymbol{x}=\boldsymbol{x}_0$, +which equals + +$$ +\boldsymbol{A}\boldsymbol{x}_0-\boldsymbol{b}, +$$ + +and +$\boldsymbol{x}_0=0$ it is equal $-\boldsymbol{b}$. + + + + +## Final expressions +We can compute the residual iteratively as + +$$ +\boldsymbol{r}_{k+1}=\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_{k+1}, +$$ + +which equals + +$$ +\boldsymbol{b}-\boldsymbol{A}(\boldsymbol{x}_k+\alpha_k\boldsymbol{r}_k), +$$ + +or + +$$ +(\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_k)-\alpha_k\boldsymbol{A}\boldsymbol{r}_k, +$$ + +which gives + +$$ +\alpha_k = \frac{\boldsymbol{r}_k^T\boldsymbol{r}_k}{\boldsymbol{r}_k^T\boldsymbol{A}\boldsymbol{r}_k} +$$ + +leading to the iterative scheme + +$$ +\boldsymbol{x}_{k+1}=\boldsymbol{x}_k-\alpha_k\boldsymbol{r}_{k}, +$$ + +## Steepest descent example + +import numpy as np +import numpy.linalg as la + +import scipy.optimize as sopt + +import matplotlib.pyplot as pt +from mpl_toolkits.mplot3d import axes3d + +def f(x): + return 0.5*x[0]**2 + 2.5*x[1]**2 + +def df(x): + return np.array([x[0], 5*x[1]]) + +fig = pt.figure() +ax = fig.gca(projection="3d") + +xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j] +fmesh = f(np.array([xmesh, ymesh])) +ax.plot_surface(xmesh, ymesh, fmesh) + +And then as countor plot + +pt.axis("equal") +pt.contour(xmesh, ymesh, fmesh) +guesses = [np.array([2, 2./5])] + +Find guesses + +x = guesses[-1] +s = -df(x) + +Run it! + +def f1d(alpha): + return f(x + alpha*s) + +alpha_opt = sopt.golden(f1d) +next_guess = x + alpha_opt * s +guesses.append(next_guess) +print(next_guess) + +What happened? + +pt.axis("equal") +pt.contour(xmesh, ymesh, fmesh, 50) +it_array = np.array(guesses) +pt.plot(it_array.T[0], it_array.T[1], "x-") + +## Conjugate gradient method +In the CG method we define so-called conjugate directions and two vectors +$\boldsymbol{s}$ and $\boldsymbol{t}$ +are said to be +conjugate if + +$$ +\boldsymbol{s}^T\boldsymbol{A}\boldsymbol{t}= 0. +$$ + +The philosophy of the CG method is to perform searches in various conjugate directions +of our vectors $\boldsymbol{x}_i$ obeying the above criterion, namely + +$$ +\boldsymbol{x}_i^T\boldsymbol{A}\boldsymbol{x}_j= 0. +$$ + +Two vectors are conjugate if they are orthogonal with respect to +this inner product. Being conjugate is a symmetric relation: if $\boldsymbol{s}$ is conjugate to $\boldsymbol{t}$, then $\boldsymbol{t}$ is conjugate to $\boldsymbol{s}$. + + + + +## Conjugate gradient method +An example is given by the eigenvectors of the matrix + +$$ +\boldsymbol{v}_i^T\boldsymbol{A}\boldsymbol{v}_j= \lambda\boldsymbol{v}_i^T\boldsymbol{v}_j, +$$ + +which is zero unless $i=j$. + + + + + +## Conjugate gradient method +Assume now that we have a symmetric positive-definite matrix $\boldsymbol{A}$ of size +$n\times n$. At each iteration $i+1$ we obtain the conjugate direction of a vector + +$$ +\boldsymbol{x}_{i+1}=\boldsymbol{x}_{i}+\alpha_i\boldsymbol{p}_{i}. +$$ + +We assume that $\boldsymbol{p}_{i}$ is a sequence of $n$ mutually conjugate directions. +Then the $\boldsymbol{p}_{i}$ form a basis of $R^n$ and we can expand the solution +$ \boldsymbol{A}\boldsymbol{x} = \boldsymbol{b}$ in this basis, namely + +$$ +\boldsymbol{x} = \sum^{n}_{i=1} \alpha_i \boldsymbol{p}_i. +$$ + +## Conjugate gradient method +The coefficients are given by + +$$ +\mathbf{A}\mathbf{x} = \sum^{n}_{i=1} \alpha_i \mathbf{A} \mathbf{p}_i = \mathbf{b}. +$$ + +Multiplying with $\boldsymbol{p}_k^T$ from the left gives + +$$ +\boldsymbol{p}_k^T \boldsymbol{A}\boldsymbol{x} = \sum^{n}_{i=1} \alpha_i\boldsymbol{p}_k^T \boldsymbol{A}\boldsymbol{p}_i= \boldsymbol{p}_k^T \boldsymbol{b}, +$$ + +and we can define the coefficients $\alpha_k$ as + +$$ +\alpha_k = \frac{\boldsymbol{p}_k^T \boldsymbol{b}}{\boldsymbol{p}_k^T \boldsymbol{A} \boldsymbol{p}_k} +$$ + +## Conjugate gradient method and iterations + +If we choose the conjugate vectors $\boldsymbol{p}_k$ carefully, +then we may not need all of them to obtain a good approximation to the solution +$\boldsymbol{x}$. +We want to regard the conjugate gradient method as an iterative method. +This will us to solve systems where $n$ is so large that the direct +method would take too much time. + +We denote the initial guess for $\boldsymbol{x}$ as $\boldsymbol{x}_0$. +We can assume without loss of generality that + +$$ +\boldsymbol{x}_0=0, +$$ + +or consider the system + +$$ +\boldsymbol{A}\boldsymbol{z} = \boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_0, +$$ + +instead. + + + + + +## Conjugate gradient method +One can show that the solution $\boldsymbol{x}$ is also the unique minimizer of the quadratic form + +$$ +f(\boldsymbol{x}) = \frac{1}{2}\boldsymbol{x}^T\boldsymbol{A}\boldsymbol{x} - \boldsymbol{x}^T \boldsymbol{x} , \quad \boldsymbol{x}\in\mathbf{R}^n. +$$ + +This suggests taking the first basis vector $\boldsymbol{p}_1$ +to be the gradient of $f$ at $\boldsymbol{x}=\boldsymbol{x}_0$, +which equals + +$$ +\boldsymbol{A}\boldsymbol{x}_0-\boldsymbol{b}, +$$ + +and +$\boldsymbol{x}_0=0$ it is equal $-\boldsymbol{b}$. +The other vectors in the basis will be conjugate to the gradient, +hence the name conjugate gradient method. + + + + + +## Conjugate gradient method +Let $\boldsymbol{r}_k$ be the residual at the $k$-th step: + +$$ +\boldsymbol{r}_k=\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_k. +$$ + +Note that $\boldsymbol{r}_k$ is the negative gradient of $f$ at +$\boldsymbol{x}=\boldsymbol{x}_k$, +so the gradient descent method would be to move in the direction $\boldsymbol{r}_k$. +Here, we insist that the directions $\boldsymbol{p}_k$ are conjugate to each other, +so we take the direction closest to the gradient $\boldsymbol{r}_k$ +under the conjugacy constraint. +This gives the following expression + +$$ +\boldsymbol{p}_{k+1}=\boldsymbol{r}_k-\frac{\boldsymbol{p}_k^T \boldsymbol{A}\boldsymbol{r}_k}{\boldsymbol{p}_k^T\boldsymbol{A}\boldsymbol{p}_k} \boldsymbol{p}_k. +$$ + +## Conjugate gradient method +We can also compute the residual iteratively as + +$$ +\boldsymbol{r}_{k+1}=\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_{k+1}, +$$ + +which equals + +$$ +\boldsymbol{b}-\boldsymbol{A}(\boldsymbol{x}_k+\alpha_k\boldsymbol{p}_k), +$$ + +or + +$$ +(\boldsymbol{b}-\boldsymbol{A}\boldsymbol{x}_k)-\alpha_k\boldsymbol{A}\boldsymbol{p}_k, +$$ + +which gives + +$$ +\boldsymbol{r}_{k+1}=\boldsymbol{r}_k-\boldsymbol{A}\boldsymbol{p}_{k}, +$$ + +## Revisiting our first homework + +We will use linear regression as a case study for the gradient descent +methods. Linear regression is a great test case for the gradient +descent methods discussed in the lectures since it has several +desirable properties such as: + +1. An analytical solution (recall homework set 1). + +2. The gradient can be computed analytically. + +3. The cost function is convex which guarantees that gradient descent converges for small enough learning rates + +We revisit an example similar to what we had in the first homework set. We had a function of the type + +x = 2*np.random.rand(m,1) +y = 4+3*x+np.random.randn(m,1) + +with $x_i \in [0,1] $ is chosen randomly using a uniform distribution. Additionally we have a stochastic noise chosen according to a normal distribution $\cal {N}(0,1)$. +The linear regression model is given by + +$$ +h_\beta(x) = \boldsymbol{y} = \beta_0 + \beta_1 x, +$$ + +such that + +$$ +\boldsymbol{y}_i = \beta_0 + \beta_1 x_i. +$$ + +## Gradient descent example + +Let $\mathbf{y} = (y_1,\cdots,y_n)^T$, $\mathbf{\boldsymbol{y}} = (\boldsymbol{y}_1,\cdots,\boldsymbol{y}_n)^T$ and $\beta = (\beta_0, \beta_1)^T$ + +It is convenient to write $\mathbf{\boldsymbol{y}} = X\beta$ where $X \in \mathbb{R}^{100 \times 2} $ is the design matrix given by (we keep the intercept here) + +$$ +X \equiv \begin{bmatrix} +1 & x_1 \\ +\vdots & \vdots \\ +1 & x_{100} & \\ +\end{bmatrix}. +$$ + +The cost/loss/risk function is given by ( + +$$ +C(\beta) = \frac{1}{n}||X\beta-\mathbf{y}||_{2}^{2} = \frac{1}{n}\sum_{i=1}^{100}\left[ (\beta_0 + \beta_1 x_i)^2 - 2 y_i (\beta_0 + \beta_1 x_i) + y_i^2\right] +$$ + +and we want to find $\beta$ such that $C(\beta)$ is minimized. + + +## The derivative of the cost/loss function + +Computing $\partial C(\beta) / \partial \beta_0$ and $\partial C(\beta) / \partial \beta_1$ we can show that the gradient can be written as + +$$ +\nabla_{\beta} C(\beta) = \frac{2}{n}\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\ +\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\ +\end{bmatrix} = \frac{2}{n}X^T(X\beta - \mathbf{y}), +$$ + +where $X$ is the design matrix defined above. + + +## The Hessian matrix +The Hessian matrix of $C(\beta)$ is given by + +$$ +\boldsymbol{H} \equiv \begin{bmatrix} +\frac{\partial^2 C(\beta)}{\partial \beta_0^2} & \frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} \\ +\frac{\partial^2 C(\beta)}{\partial \beta_0 \partial \beta_1} & \frac{\partial^2 C(\beta)}{\partial \beta_1^2} & \\ +\end{bmatrix} = \frac{2}{n}X^T X. +$$ + +This result implies that $C(\beta)$ is a convex function since the matrix $X^T X$ always is positive semi-definite. + + + + + +## Simple program + +We can now write a program that minimizes $C(\beta)$ using the gradient descent method with a constant learning rate $\gamma$ according to + +$$ +\beta_{k+1} = \beta_k - \gamma \nabla_\beta C(\beta_k), \ k=0,1,\cdots +$$ + +We can use the expression we computed for the gradient and let use a +$\beta_0$ be chosen randomly and let $\gamma = 0.001$. Stop iterating +when $||\nabla_\beta C(\beta_k) || \leq \epsilon = 10^{-8}$. **Note that the code below does not include the latter stop criterion**. + +And finally we can compare our solution for $\beta$ with the analytic result given by +$\beta= (X^TX)^{-1} X^T \mathbf{y}$. + + +## Gradient Descent Example + +Here our simple example + + +# Importing various packages +from random import random, seed +import numpy as np +import matplotlib.pyplot as plt +from mpl_toolkits.mplot3d import Axes3D +from matplotlib import cm +from matplotlib.ticker import LinearLocator, FormatStrFormatter +import sys + +# the number of datapoints +n = 100 +x = 2*np.random.rand(n,1) +y = 4+3*x+np.random.randn(n,1) + +X = np.c_[np.ones((n,1)), x] +# Hessian matrix +H = (2.0/n)* X.T @ X +# Get the eigenvalues +EigValues, EigVectors = np.linalg.eig(H) +print(EigValues) + +beta_linreg = np.linalg.inv(X.T @ X) @ X.T @ y +print(beta_linreg) +beta = np.random.randn(2,1) + +eta = 1.0/np.max(EigValues) +Niterations = 1000 + +for iter in range(Niterations): + gradient = (2.0/n)*X.T @ (X @ beta-y) + beta -= eta*gradient + +print(beta) +xnew = np.array([[0],[2]]) +xbnew = np.c_[np.ones((2,1)), xnew] +ypredict = xbnew.dot(beta) +ypredict2 = xbnew.dot(beta_linreg) +plt.plot(xnew, ypredict, "r-") +plt.plot(xnew, ypredict2, "b-") +plt.plot(x, y ,'ro') +plt.axis([0,2.0,0, 15.0]) +plt.xlabel(r'$x$') +plt.ylabel(r'$y$') +plt.title(r'Gradient descent example') +plt.show() + +## And a corresponding example using **scikit-learn** + +# Importing various packages +from random import random, seed +import numpy as np +import matplotlib.pyplot as plt +from sklearn.linear_model import SGDRegressor + +n = 100 +x = 2*np.random.rand(n,1) +y = 4+3*x+np.random.randn(n,1) + +X = np.c_[np.ones((n,1)), x] +beta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y) +print(beta_linreg) +sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1) +sgdreg.fit(x,y.ravel()) +print(sgdreg.intercept_, sgdreg.coef_) + +## Gradient descent and Ridge + +We have also discussed Ridge regression where the loss function contains a regularized term given by the $L_2$ norm of $\beta$, + +$$ +C_{\text{ridge}}(\beta) = \frac{1}{n}||X\beta -\mathbf{y}||^2 + \lambda ||\beta||^2, \ \lambda \geq 0. +$$ + +In order to minimize $C_{\text{ridge}}(\beta)$ using GD we only have adjust the gradient as follows + +$$ +\nabla_\beta C_{\text{ridge}}(\beta) = \frac{2}{n}\begin{bmatrix} \sum_{i=1}^{100} \left(\beta_0+\beta_1x_i-y_i\right) \\ +\sum_{i=1}^{100}\left( x_i (\beta_0+\beta_1x_i)-y_ix_i\right) \\ +\end{bmatrix} + 2\lambda\begin{bmatrix} \beta_0 \\ \beta_1\end{bmatrix} = 2 (X^T(X\beta - \mathbf{y})+\lambda \beta). +$$ + +We can easily extend our program to minimize $C_{\text{ridge}}(\beta)$ using gradient descent and compare with the analytical solution given by + +$$ +\beta_{\text{ridge}} = \left(X^T X + \lambda I_{2 \times 2} \right)^{-1} X^T \mathbf{y}. +$$ + +## Program example for gradient descent with Ridge Regression + +from random import random, seed +import numpy as np +import matplotlib.pyplot as plt +from mpl_toolkits.mplot3d import Axes3D +from matplotlib import cm +from matplotlib.ticker import LinearLocator, FormatStrFormatter +import sys + +# the number of datapoints +n = 100 +x = 2*np.random.rand(n,1) +y = 4+3*x+np.random.randn(n,1) + +X = np.c_[np.ones((n,1)), x] +XT_X = X.T @ X + +#Ridge parameter lambda +lmbda = 0.001 +Id = lmbda* np.eye(XT_X.shape[0]) + +beta_linreg = np.linalg.inv(XT_X+Id) @ X.T @ y +print(beta_linreg) +# Start plain gradient descent +beta = np.random.randn(2,1) + +eta = 0.1 +Niterations = 100 + +for iter in range(Niterations): + gradients = 2.0/n*X.T @ (X @ (beta)-y)+2*lmbda*beta + beta -= eta*gradients + +print(beta) +ypredict = X @ beta +ypredict2 = X @ beta_linreg +plt.plot(x, ypredict, "r-") +plt.plot(x, ypredict2, "b-") +plt.plot(x, y ,'ro') +plt.axis([0,2.0,0, 15.0]) +plt.xlabel(r'$x$') +plt.ylabel(r'$y$') +plt.title(r'Gradient descent example for Ridge') +plt.show() + +## Using gradient descent methods, limitations + +* **Gradient descent (GD) finds local minima of our function**. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our cost/loss/risk function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance. + +* **GD is sensitive to initial conditions**. One consequence of the local nature of GD is that initial conditions matter. Depending on where one starts, one will end up at a different local minima. Therefore, it is very important to think about how one initializes the training process. This is true for GD as well as more complicated variants of GD. + +* **Gradients are computationally expensive to calculate for large datasets**. In many cases in statistics and ML, the cost/loss/risk function is a sum of terms, with one term for each data point. For example, in linear regression, $E \propto \sum_{i=1}^n (y_i - \mathbf{w}^T\cdot\mathbf{x}_i)^2$; for logistic regression, the square error is replaced by the cross entropy. To calculate the gradient we have to sum over *all* $n$ data points. Doing this at every GD step becomes extremely computationally expensive. An ingenious solution to this, is to calculate the gradients using small subsets of the data called "mini batches". This has the added benefit of introducing stochasticity into our algorithm. + +* **GD is very sensitive to choices of learning rates**. GD is extremely sensitive to the choice of learning rates. If the learning rate is very small, the training process take an extremely long time. For larger learning rates, GD can diverge and give poor results. Furthermore, depending on what the local landscape looks like, we have to modify the learning rates to ensure convergence. Ideally, we would *adaptively* choose the learning rates to match the landscape. + +* **GD treats all directions in parameter space uniformly.** Another major drawback of GD is that unlike Newton's method, the learning rate for GD is the same in all directions in parameter space. For this reason, the maximum learning rate is set by the behavior of the steepest direction and this can significantly slow down training. Ideally, we would like to take large steps in flat directions and small steps in steep directions. Since we are exploring rugged landscapes where curvatures change, this requires us to keep track of not only the gradient but second derivatives. The ideal scenario would be to calculate the Hessian but this proves to be too computationally expensive. + +* GD can take exponential time to escape saddle points, even with random initialization. As we mentioned, GD is extremely sensitive to initial condition since it determines the particular local minimum GD would eventually reach. However, even with a good initialization scheme, through the introduction of randomness, GD can still take exponential time to escape saddle points. + +## Stochastic Gradient Descent + +Stochastic gradient descent (SGD) and variants thereof address some of +the shortcomings of the Gradient descent method discussed above. + +The underlying idea of SGD comes from the observation that the cost +function, which we want to minimize, can almost always be written as a +sum over $n$ data points $\{\mathbf{x}_i\}_{i=1}^n$, + +$$ +C(\mathbf{\beta}) = \sum_{i=1}^n c_i(\mathbf{x}_i, +\mathbf{\beta}). +$$ + +## Computation of gradients + +This in turn means that the gradient can be +computed as a sum over $i$-gradients + +$$ +\nabla_\beta C(\mathbf{\beta}) = \sum_i^n \nabla_\beta c_i(\mathbf{x}_i, +\mathbf{\beta}). +$$ + +Stochasticity/randomness is introduced by only taking the +gradient on a subset of the data called minibatches. If there are $n$ +data points and the size of each minibatch is $M$, there will be $n/M$ +minibatches. We denote these minibatches by $B_k$ where +$k=1,\cdots,n/M$. + + +## SGD example +As an example, suppose we have $10$ data points $(\mathbf{x}_1,\cdots, \mathbf{x}_{10})$ +and we choose to have $M=5$ minibathces, +then each minibatch contains two data points. In particular we have +$B_1 = (\mathbf{x}_1,\mathbf{x}_2), \cdots, B_5 = +(\mathbf{x}_9,\mathbf{x}_{10})$. Note that if you choose $M=1$ you +have only a single batch with all data points and on the other extreme, +you may choose $M=n$ resulting in a minibatch for each datapoint, i.e +$B_k = \mathbf{x}_k$. + +The idea is now to approximate the gradient by replacing the sum over +all data points with a sum over the data points in one the minibatches +picked at random in each gradient descent step + +$$ +\nabla_{\beta} +C(\mathbf{\beta}) = \sum_{i=1}^n \nabla_\beta c_i(\mathbf{x}_i, +\mathbf{\beta}) \rightarrow \sum_{i \in B_k}^n \nabla_\beta +c_i(\mathbf{x}_i, \mathbf{\beta}). +$$ + +## The gradient step + +Thus a gradient descent step now looks like + +$$ +\beta_{j+1} = \beta_j - \gamma_j \sum_{i \in B_k}^n \nabla_\beta c_i(\mathbf{x}_i, +\mathbf{\beta}) +$$ + +where $k$ is picked at random with equal +probability from $[1,n/M]$. An iteration over the number of +minibathces (n/M) is commonly referred to as an epoch. Thus it is +typical to choose a number of epochs and for each epoch iterate over +the number of minibatches, as exemplified in the code below. + + +## Simple example code + +import numpy as np + +n = 100 #100 datapoints +M = 5 #size of each minibatch +m = int(n/M) #number of minibatches +n_epochs = 10 #number of epochs + +j = 0 +for epoch in range(1,n_epochs+1): + for i in range(m): + k = np.random.randint(m) #Pick the k-th minibatch at random + #Compute the gradient using the data in minibatch Bk + #Compute new suggestion for + j += 1 + +Taking the gradient only on a subset of the data has two important +benefits. First, it introduces randomness which decreases the chance +that our opmization scheme gets stuck in a local minima. Second, if +the size of the minibatches are small relative to the number of +datapoints ($M < n$), the computation of the gradient is much +cheaper since we sum over the datapoints in the $k-th$ minibatch and not +all $n$ datapoints. + + +## When do we stop? + +A natural question is when do we stop the search for a new minimum? +One possibility is to compute the full gradient after a given number +of epochs and check if the norm of the gradient is smaller than some +threshold and stop if true. However, the condition that the gradient +is zero is valid also for local minima, so this would only tell us +that we are close to a local/global minimum. However, we could also +evaluate the cost function at this point, store the result and +continue the search. If the test kicks in at a later stage we can +compare the values of the cost function and keep the $\beta$ that +gave the lowest value. + + +## Slightly different approach + +Another approach is to let the step length $\gamma_j$ depend on the +number of epochs in such a way that it becomes very small after a +reasonable time such that we do not move at all. + +As an example, let $e = 0,1,2,3,\cdots$ denote the current epoch and let $t_0, t_1 > 0$ be two fixed numbers. Furthermore, let $t = e \cdot m + i$ where $m$ is the number of minibatches and $i=0,\cdots,m-1$. Then the function $$\gamma_j(t; t_0, t_1) = \frac{t_0}{t+t_1} $$ goes to zero as the number of epochs gets large. I.e. we start with a step length $\gamma_j (0; t_0, t_1) = t_0/t_1$ which decays in *time* $t$. + +In this way we can fix the number of epochs, compute $\beta$ and +evaluate the cost function at the end. Repeating the computation will +give a different result since the scheme is random by design. Then we +pick the final $\beta$ that gives the lowest value of the cost +function. + +import numpy as np + +def step_length(t,t0,t1): + return t0/(t+t1) + +n = 100 #100 datapoints +M = 5 #size of each minibatch +m = int(n/M) #number of minibatches +n_epochs = 500 #number of epochs +t0 = 1.0 +t1 = 10 + +gamma_j = t0/t1 +j = 0 +for epoch in range(1,n_epochs+1): + for i in range(m): + k = np.random.randint(m) #Pick the k-th minibatch at random + #Compute the gradient using the data in minibatch Bk + #Compute new suggestion for beta + t = epoch*m+i + gamma_j = step_length(t,t0,t1) + j += 1 + +print("gamma_j after %d epochs: %g" % (n_epochs,gamma_j)) + +## Program for stochastic gradient + +# Importing various packages +from math import exp, sqrt +from random import random, seed +import numpy as np +import matplotlib.pyplot as plt +from sklearn.linear_model import SGDRegressor + +m = 100 +x = 2*np.random.rand(m,1) +y = 4+3*x+np.random.randn(m,1) + +X = np.c_[np.ones((m,1)), x] +theta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y) +print("Own inversion") +print(theta_linreg) +sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1) +sgdreg.fit(x,y.ravel()) +print("sgdreg from scikit") +print(sgdreg.intercept_, sgdreg.coef_) + + +theta = np.random.randn(2,1) +eta = 0.1 +Niterations = 1000 + + +for iter in range(Niterations): + gradients = 2.0/m*X.T @ ((X @ theta)-y) + theta -= eta*gradients +print("theta from own gd") +print(theta) + +xnew = np.array([[0],[2]]) +Xnew = np.c_[np.ones((2,1)), xnew] +ypredict = Xnew.dot(theta) +ypredict2 = Xnew.dot(theta_linreg) + + +n_epochs = 50 +t0, t1 = 5, 50 +def learning_schedule(t): + return t0/(t+t1) + +theta = np.random.randn(2,1) + +for epoch in range(n_epochs): + for i in range(m): + random_index = np.random.randint(m) + xi = X[random_index:random_index+1] + yi = y[random_index:random_index+1] + gradients = 2 * xi.T @ ((xi @ theta)-yi) + eta = learning_schedule(epoch*m+i) + theta = theta - eta*gradients +print("theta from own sdg") +print(theta) + +plt.plot(xnew, ypredict, "r-") +plt.plot(xnew, ypredict2, "b-") +plt.plot(x, y ,'ro') +plt.axis([0,2.0,0, 15.0]) +plt.xlabel(r'$x$') +plt.ylabel(r'$y$') +plt.title(r'Random numbers ') +plt.show() + +**Challenge**: try to write a similar code for a Logistic Regression case. \ No newline at end of file diff --git a/doc/LectureNotes/_toc.yml b/doc/LectureNotes/_toc.yml index b39daaede..a2cb3c3bd 100644 --- a/doc/LectureNotes/_toc.yml +++ b/doc/LectureNotes/_toc.yml @@ -11,15 +11,3 @@ - file: chapter2.ipynb - file: chapter3.ipynb - file: chapter4.ipynb -- part: Deep Learning - numbered: true - chapters: - - file: chapter5.ipynb - - file: chapter6.ipynb - - file: chapter7.ipynb - - file: chapter8.ipynb -- part: Trees and Ensemble Methods - numbered: true - chapters: - - file: chapter9.ipynb - - file: chapter10.ipynb diff --git a/doc/LectureNotes/chapter3.ipynb b/doc/LectureNotes/chapter3.ipynb new file mode 100644 index 000000000..5ede9f687 --- /dev/null +++ b/doc/LectureNotes/chapter3.ipynb @@ -0,0 +1,1378 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Ridge and Lasso Regression\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember10.mp4?vrtx=view-as-webpage)\n", + "\n", + "\n", + "## The singular value decomposition\n", + "\n", + "The examples we have looked at so far are cases where we normally can\n", + "invert the matrix $\\boldsymbol{X}^T\\boldsymbol{X}$. Using a polynomial expansion as we\n", + "did both for the masses and the fitting of the equation of state,\n", + "leads to row vectors of the design matrix which are essentially\n", + "orthogonal due to the polynomial character of our model. Obtaining the inverse of the design matrix is then often done via a so-called LU, QR or Cholesky decomposition. \n", + "\n", + "\n", + "\n", + "This may\n", + "however not the be case in general and a standard matrix inversion\n", + "algorithm based on say LU, QR or Cholesky decomposition may lead to singularities. We will see examples of this below.\n", + "\n", + "There is however a way to partially circumvent this problem and also gain some insights about the ordinary least squares approach, and later shrinkage methods like Ridge and Lasso regressions. \n", + "\n", + "This is given by the **Singular Value Decomposition** algorithm, perhaps\n", + "the most powerful linear algebra algorithm. Let us look at a\n", + "different example where we may have problems with the standard matrix\n", + "inversion algorithm. Thereafter we dive into the math of the SVD.\n", + "\n", + "\n", + "\n", + "One of the typical problems we encounter with linear regression, in particular \n", + "when the matrix $\\boldsymbol{X}$ (our so-called design matrix) is high-dimensional, \n", + "are problems with near singular or singular matrices. The column vectors of $\\boldsymbol{X}$ \n", + "may be linearly dependent, normally referred to as super-collinearity. \n", + "This means that the matrix may be rank deficient and it is basically impossible to \n", + "to model the data using linear regression. As an example, consider the matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "\\mathbf{X} & = \\left[\n", + "\\begin{array}{rrr}\n", + "1 & -1 & 2\n", + "\\\\\n", + "1 & 0 & 1\n", + "\\\\\n", + "1 & 2 & -1\n", + "\\\\\n", + "1 & 1 & 0\n", + "\\end{array} \\right]\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The columns of $\\boldsymbol{X}$ are linearly dependent. We see this easily since the \n", + "the first column is the row-wise sum of the other two columns. The rank (more correct,\n", + "the column rank) of a matrix is the dimension of the space spanned by the\n", + "column vectors. Hence, the rank of $\\mathbf{X}$ is equal to the number\n", + "of linearly independent columns. In this particular case the matrix has rank 2.\n", + "\n", + "Super-collinearity of an $(n \\times p)$-dimensional design matrix $\\mathbf{X}$ implies\n", + "that the inverse of the matrix $\\boldsymbol{X}^T\\boldsymbol{X}$ (the matrix we need to invert to solve the linear regression equations) is non-invertible. If we have a square matrix that does not have an inverse, we say this matrix singular. The example here demonstrates this" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "\\boldsymbol{X} & = \\left[\n", + "\\begin{array}{rr}\n", + "1 & -1\n", + "\\\\\n", + "1 & -1\n", + "\\end{array} \\right].\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We see easily that $\\mbox{det}(\\boldsymbol{X}) = x_{11} x_{22} - x_{12} x_{21} = 1 \\times (-1) - 1 \\times (-1) = 0$. Hence, $\\mathbf{X}$ is singular and its inverse is undefined.\n", + "This is equivalent to saying that the matrix $\\boldsymbol{X}$ has at least an eigenvalue which is zero.\n", + "\n", + "\n", + "If our design matrix $\\boldsymbol{X}$ which enters the linear regression problem" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "
\n", + "\n", + "$$\n", + "\\begin{equation}\n", + "\\boldsymbol{\\beta} = (\\boldsymbol{X}^{T} \\boldsymbol{X})^{-1} \\boldsymbol{X}^{T} \\boldsymbol{y},\n", + "\\label{_auto1} \\tag{1}\n", + "\\end{equation}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "has linearly dependent column vectors, we will not be able to compute the inverse\n", + "of $\\boldsymbol{X}^T\\boldsymbol{X}$ and we cannot find the parameters (estimators) $\\beta_i$. \n", + "The estimators are only well-defined if $(\\boldsymbol{X}^{T}\\boldsymbol{X})^{-1}$ exits. \n", + "This is more likely to happen when the matrix $\\boldsymbol{X}$ is high-dimensional. In this case it is likely to encounter a situation where \n", + "the regression parameters $\\beta_i$ cannot be estimated.\n", + "\n", + "A cheap *ad hoc* approach is simply to add a small diagonal component to the matrix to invert, that is we change" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^{T} \\boldsymbol{X} \\rightarrow \\boldsymbol{X}^{T} \\boldsymbol{X}+\\lambda \\boldsymbol{I},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\boldsymbol{I}$ is the identity matrix. When we discuss **Ridge** regression this is actually what we end up evaluating. The parameter $\\lambda$ is called a hyperparameter. More about this later. \n", + "\n", + "\n", + "\n", + "\n", + "\n", + "From standard linear algebra we know that a square matrix $\\boldsymbol{X}$ can be diagonalized if and only it is \n", + "a so-called [normal matrix](https://en.wikipedia.org/wiki/Normal_matrix), that is if $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times n}$\n", + "we have $\\boldsymbol{X}\\boldsymbol{X}^T=\\boldsymbol{X}^T\\boldsymbol{X}$ or if $\\boldsymbol{X}\\in {\\mathbb{C}}^{n\\times n}$ we have $\\boldsymbol{X}\\boldsymbol{X}^{\\dagger}=\\boldsymbol{X}^{\\dagger}\\boldsymbol{X}$.\n", + "The matrix has then a set of eigenpairs" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\lambda_1,\\boldsymbol{u}_1),\\dots, (\\lambda_n,\\boldsymbol{u}_n),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and the eigenvalues are given by the diagonal matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\Sigma}=\\mathrm{Diag}(\\lambda_1, \\dots,\\lambda_n).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The matrix $\\boldsymbol{X}$ can be written in terms of an orthogonal/unitary transformation $\\boldsymbol{U}$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{U}\\boldsymbol{U}^T=\\boldsymbol{I}$ or $\\boldsymbol{U}\\boldsymbol{U}^{\\dagger}=\\boldsymbol{I}$.\n", + "\n", + "Not all square matrices are diagonalizable. A matrix like the one discussed above" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\begin{bmatrix} \n", + "1& -1 \\\\\n", + "1& -1\\\\\n", + "\\end{bmatrix}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "is not diagonalizable, it is a so-called [defective matrix](https://en.wikipedia.org/wiki/Defective_matrix). It is easy to see that the condition\n", + "$\\boldsymbol{X}\\boldsymbol{X}^T=\\boldsymbol{X}^T\\boldsymbol{X}$ is not fulfilled. \n", + "\n", + "\n", + "\n", + "## The SVD, a Fantastic Algorithm\n", + "\n", + "\n", + "However, and this is the strength of the SVD algorithm, any general\n", + "matrix $\\boldsymbol{X}$ can be decomposed in terms of a diagonal matrix and\n", + "two orthogonal/unitary matrices. The [Singular Value Decompostion\n", + "(SVD) theorem](https://en.wikipedia.org/wiki/Singular_value_decomposition)\n", + "states that a general $m\\times n$ matrix $\\boldsymbol{X}$ can be written in\n", + "terms of a diagonal matrix $\\boldsymbol{\\Sigma}$ of dimensionality $m\\times n$\n", + "and two orthognal matrices $\\boldsymbol{U}$ and $\\boldsymbol{V}$, where the first has\n", + "dimensionality $m \\times m$ and the last dimensionality $n\\times n$.\n", + "We have then" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "As an example, the above defective matrix can be decomposed as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\frac{1}{\\sqrt{2}}\\begin{bmatrix} 1& 1 \\\\ 1& -1\\\\ \\end{bmatrix} \\begin{bmatrix} 2& 0 \\\\ 0& 0\\\\ \\end{bmatrix} \\frac{1}{\\sqrt{2}}\\begin{bmatrix} 1& -1 \\\\ 1& 1\\\\ \\end{bmatrix}=\\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with eigenvalues $\\sigma_1=2$ and $\\sigma_2=0$. \n", + "The SVD exits always! \n", + "\n", + "The SVD\n", + "decomposition (singular values) gives eigenvalues \n", + "$\\sigma_i\\geq\\sigma_{i+1}$ for all $i$ and for dimensions larger than $i=p$, the\n", + "eigenvalues (singular values) are zero.\n", + "\n", + "In the general case, where our design matrix $\\boldsymbol{X}$ has dimension\n", + "$n\\times p$, the matrix is thus decomposed into an $n\\times n$\n", + "orthogonal matrix $\\boldsymbol{U}$, a $p\\times p$ orthogonal matrix $\\boldsymbol{V}$\n", + "and a diagonal matrix $\\boldsymbol{\\Sigma}$ with $r=\\mathrm{min}(n,p)$\n", + "singular values $\\sigma_i\\geq 0$ on the main diagonal and zeros filling\n", + "the rest of the matrix. There are at most $p$ singular values\n", + "assuming that $n > p$. In our regression examples for the nuclear\n", + "masses and the equation of state this is indeed the case, while for\n", + "the Ising model we have $p > n$. These are often cases that lead to\n", + "near singular or singular matrices.\n", + "\n", + "The columns of $\\boldsymbol{U}$ are called the left singular vectors while the columns of $\\boldsymbol{V}$ are the right singular vectors.\n", + "\n", + "## Economy-size SVD\n", + "\n", + "If we assume that $n > p$, then our matrix $\\boldsymbol{U}$ has dimension $n\n", + "\\times n$. The last $n-p$ columns of $\\boldsymbol{U}$ become however\n", + "irrelevant in our calculations since they are multiplied with the\n", + "zeros in $\\boldsymbol{\\Sigma}$.\n", + "\n", + "The economy-size decomposition removes extra rows or columns of zeros\n", + "from the diagonal matrix of singular values, $\\boldsymbol{\\Sigma}$, along with the columns\n", + "in either $\\boldsymbol{U}$ or $\\boldsymbol{V}$ that multiply those zeros in the expression. \n", + "Removing these zeros and columns can improve execution time\n", + "and reduce storage requirements without compromising the accuracy of\n", + "the decomposition.\n", + "\n", + "If $n > p$, we keep only the first $p$ columns of $\\boldsymbol{U}$ and $\\boldsymbol{\\Sigma}$ has dimension $p\\times p$. \n", + "If $p > n$, then only the first $n$ columns of $\\boldsymbol{V}$ are computed and $\\boldsymbol{\\Sigma}$ has dimension $n\\times n$.\n", + "The $n=p$ case is obvious, we retain the full SVD. \n", + "In general the economy-size SVD leads to less FLOPS and still conserving the desired accuracy." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "# SVD inversion\n", + "def SVDinv(A):\n", + " ''' Takes as input a numpy matrix A and returns inv(A) based on singular value decomposition (SVD).\n", + " SVD is numerically more stable than the inversion algorithms provided by\n", + " numpy and scipy.linalg at the cost of being slower.\n", + " '''\n", + " U, s, VT = np.linalg.svd(A)\n", + "# print('test U')\n", + "# print( (np.transpose(U) @ U - U @np.transpose(U)))\n", + "# print('test VT')\n", + "# print( (np.transpose(VT) @ VT - VT @np.transpose(VT)))\n", + " print(U)\n", + " print(s)\n", + " print(VT)\n", + "\n", + " D = np.zeros((len(U),len(VT)))\n", + " for i in range(0,len(VT)):\n", + " D[i,i]=s[i]\n", + " UT = np.transpose(U); V = np.transpose(VT); invD = np.linalg.inv(D)\n", + " return np.matmul(V,np.matmul(invD,UT))\n", + "\n", + "\n", + "X = np.array([ [1.0, -1.0, 2.0], [1.0, 0.0, 1.0], [1.0, 2.0, -1.0], [1.0, 1.0, 0.0] ])\n", + "print(X)\n", + "A = np.transpose(X) @ X\n", + "print(A)\n", + "# Brute force inversion of super-collinear matrix\n", + "#B = np.linalg.inv(A)\n", + "#print(B)\n", + "C = SVDinv(A)\n", + "print(C)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The matrix $\\boldsymbol{X}$ has columns that are linearly dependent. The first\n", + "column is the row-wise sum of the other two columns. The rank of a\n", + "matrix (the column rank) is the dimension of space spanned by the\n", + "column vectors. The rank of the matrix is the number of linearly\n", + "independent columns, in this case just $2$. We see this from the\n", + "singular values when running the above code. Running the standard\n", + "inversion algorithm for matrix inversion with $\\boldsymbol{X}^T\\boldsymbol{X}$ results\n", + "in the program terminating due to a singular matrix.\n", + "\n", + "\n", + "\n", + "\n", + "There are several interesting mathematical properties which will be\n", + "relevant when we are going to discuss the differences between say\n", + "ordinary least squares (OLS) and **Ridge** regression.\n", + "\n", + "We have from OLS that the parameters of the linear approximation are given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\tilde{y}} = \\boldsymbol{X}\\boldsymbol{\\beta} = \\boldsymbol{X}\\left(\\boldsymbol{X}^T\\boldsymbol{X}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The matrix to invert can be rewritten in terms of our SVD decomposition as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{X} = \\boldsymbol{V}\\boldsymbol{\\Sigma}^T\\boldsymbol{U}^T\\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Using the orthogonality properties of $\\boldsymbol{U}$ we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{X} = \\boldsymbol{V}\\boldsymbol{\\Sigma}^T\\boldsymbol{\\Sigma}\\boldsymbol{V}^T = \\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{D}$ being a diagonal matrix with values along the diagonal given by the singular values squared. \n", + "\n", + "This means that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{X}^T\\boldsymbol{X})\\boldsymbol{V} = \\boldsymbol{V}\\boldsymbol{D},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "that is the eigenvectors of $(\\boldsymbol{X}^T\\boldsymbol{X})$ are given by the columns of the right singular matrix of $\\boldsymbol{X}$ and the eigenvalues are the squared singular values. It is easy to show (show this) that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{X}\\boldsymbol{X}^T)\\boldsymbol{U} = \\boldsymbol{U}\\boldsymbol{D},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "that is, the eigenvectors of $(\\boldsymbol{X}\\boldsymbol{X})^T$ are the columns of the left singular matrix and the eigenvalues are the same. \n", + "\n", + "Going back to our OLS equation we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}\\boldsymbol{\\beta} = \\boldsymbol{X}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T \\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}=\\boldsymbol{U\\Sigma V^T}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T \\right)^{-1}(\\boldsymbol{U\\Sigma V^T})^T\\boldsymbol{y}=\\boldsymbol{U}\\boldsymbol{U}^T\\boldsymbol{y}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We will come back to this expression when we discuss Ridge regression. \n", + "\n", + "\n", + "$$ \\tilde{y}^{OLS}=\\boldsymbol{X}\\hat{\\beta}^{OLS}=\\sum_{j=1}^p \\boldsymbol{u}_j\\boldsymbol{u}_j^T\\boldsymbol{y}$$ and for Ridge we have \n", + "\n", + "$$ \\tilde{y}^{Ridge}=\\boldsymbol{X}\\hat{\\beta}^{Ridge}=\\sum_{j=1}^p \\boldsymbol{u}_j\\frac{\\sigma_j^2}{\\sigma_j^2+\\lambda}\\boldsymbol{u}_j^T\\boldsymbol{y}$$ . \n", + "\n", + "It is indeed the economy-sized SVD, note the summation runs up tp $$p$$ only and not $$n$$. \n", + "\n", + "Here we have that $$\\boldsymbol{X} = \\boldsymbol{U}\\boldsymbol{\\Sigma}\\boldsymbol{V}^T$$, with $$\\Sigma$$ being an $$ n\\times p$$ matrix and $$\\boldsymbol{V}$$ being a $$ p\\times p$$ matrix. We also have assumed here that $$ n > p$$. \n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Ridge and LASSO Regression\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember11.mp4?vrtx=view-as-webpage)\n", + "\n", + "Let us remind ourselves about the expression for the standard Mean Squared Error (MSE) which we used to define our cost function and the equations for the ordinary least squares (OLS) method, that is \n", + "our optimization problem is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in {\\mathbb{R}}^{p}}}\\frac{1}{n}\\left\\{\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)^T\\left(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\right)\\right\\}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or we can state it as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in\n", + "{\\mathbb{R}}^{p}}}\\frac{1}{n}\\sum_{i=0}^{n-1}\\left(y_i-\\tilde{y}_i\\right)^2=\\frac{1}{n}\\vert\\vert \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\vert\\vert_2^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have used the definition of a norm-2 vector, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\vert\\vert \\boldsymbol{x}\\vert\\vert_2 = \\sqrt{\\sum_i x_i^2}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "By minimizing the above equation with respect to the parameters\n", + "$\\boldsymbol{\\beta}$ we could then obtain an analytical expression for the\n", + "parameters $\\boldsymbol{\\beta}$. We can add a regularization parameter $\\lambda$ by\n", + "defining a new cost function to be optimized, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in\n", + "{\\mathbb{R}}^{p}}}\\frac{1}{n}\\vert\\vert \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\vert\\vert_2^2+\\lambda\\vert\\vert \\boldsymbol{\\beta}\\vert\\vert_2^2\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which leads to the Ridge regression minimization problem where we\n", + "require that $\\vert\\vert \\boldsymbol{\\beta}\\vert\\vert_2^2\\le t$, where $t$ is\n", + "a finite number larger than zero. By defining" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{X},\\boldsymbol{\\beta})=\\frac{1}{n}\\vert\\vert \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\vert\\vert_2^2+\\lambda\\vert\\vert \\boldsymbol{\\beta}\\vert\\vert_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we have a new optimization equation" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\displaystyle \\min_{\\boldsymbol{\\beta}\\in\n", + "{\\mathbb{R}}^{p}}}\\frac{1}{n}\\vert\\vert \\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta}\\vert\\vert_2^2+\\lambda\\vert\\vert \\boldsymbol{\\beta}\\vert\\vert_1\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which leads to Lasso regression. Lasso stands for least absolute shrinkage and selection operator. \n", + "\n", + "Here we have defined the norm-1 as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\vert\\vert \\boldsymbol{x}\\vert\\vert_1 = \\sum_i \\vert x_i\\vert.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Using the matrix-vector expression for Ridge regression," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\boldsymbol{X},\\boldsymbol{\\beta})=\\frac{1}{n}\\left\\{(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta})^T(\\boldsymbol{y}-\\boldsymbol{X}\\boldsymbol{\\beta})\\right\\}+\\lambda\\boldsymbol{\\beta}^T\\boldsymbol{\\beta},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "by taking the derivatives with respect to $\\boldsymbol{\\beta}$ we obtain then\n", + "a slightly modified matrix inversion problem which for finite values\n", + "of $\\lambda$ does not suffer from singularity problems. We obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{Ridge}} = \\left(\\boldsymbol{X}^T\\boldsymbol{X}+\\lambda\\boldsymbol{I}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{I}$ being a $p\\times p$ identity matrix with the constraint that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\sum_{i=0}^{p-1} \\beta_i^2 \\leq t,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $t$ a finite positive number. \n", + "\n", + "We see that Ridge regression is nothing but the standard\n", + "OLS with a modified diagonal term added to $\\boldsymbol{X}^T\\boldsymbol{X}$. The\n", + "consequences, in particular for our discussion of the bias-variance tradeoff \n", + "are rather interesting.\n", + "\n", + "Furthermore, if we use the result above in terms of the SVD decomposition (our analysis was done for the OLS method), we had" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{X}\\boldsymbol{X}^T)\\boldsymbol{U} = \\boldsymbol{U}\\boldsymbol{D}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can analyse the OLS solutions in terms of the eigenvectors (the columns) of the right singular value matrix $\\boldsymbol{U}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}\\boldsymbol{\\beta} = \\boldsymbol{X}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T \\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}=\\boldsymbol{U\\Sigma V^T}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T \\right)^{-1}(\\boldsymbol{U\\Sigma V^T})^T\\boldsymbol{y}=\\boldsymbol{U}\\boldsymbol{U}^T\\boldsymbol{y}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "For Ridge regression this becomes" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}\\boldsymbol{\\beta}^{\\mathrm{Ridge}} = \\boldsymbol{U\\Sigma V^T}\\left(\\boldsymbol{V}\\boldsymbol{D}\\boldsymbol{V}^T+\\lambda\\boldsymbol{I} \\right)^{-1}(\\boldsymbol{U\\Sigma V^T})^T\\boldsymbol{y}=\\sum_{j=0}^{p-1}\\boldsymbol{u}_j\\boldsymbol{u}_j^T\\frac{\\sigma_j^2}{\\sigma_j^2+\\lambda}\\boldsymbol{y},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with the vectors $\\boldsymbol{u}_j$ being the columns of $\\boldsymbol{U}$. \n", + "\n", + "\n", + "Since $\\lambda \\geq 0$, it means that compared to OLS, we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\sigma_j^2}{\\sigma_j^2+\\lambda} \\leq 1.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Ridge regression finds the coordinates of $\\boldsymbol{y}$ with respect to the\n", + "orthonormal basis $\\boldsymbol{U}$, it then shrinks the coordinates by\n", + "$\\frac{\\sigma_j^2}{\\sigma_j^2+\\lambda}$. Recall that the SVD has\n", + "eigenvalues ordered in a descending way, that is $\\sigma_i \\geq\n", + "\\sigma_{i+1}$.\n", + "\n", + "For small eigenvalues $\\sigma_i$ it means that their contributions become less important, a fact which can be used to reduce the number of degrees of freedom.\n", + "Actually, calculating the variance of $\\boldsymbol{X}\\boldsymbol{v}_j$ shows that this quantity is equal to $\\sigma_j^2/n$.\n", + "With a parameter $\\lambda$ we can thus shrink the role of specific parameters. \n", + "\n", + "\n", + "\n", + "For the sake of simplicity, let us assume that the design matrix is orthonormal, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}^T\\boldsymbol{X}=(\\boldsymbol{X}^T\\boldsymbol{X})^{-1} =\\boldsymbol{I}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In this case the standard OLS results in" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{OLS}} = \\boldsymbol{X}^T\\boldsymbol{y}=\\sum_{i=0}^{p-1}\\boldsymbol{u}_j\\boldsymbol{u}_j^T\\boldsymbol{y},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{Ridge}} = \\left(\\boldsymbol{I}+\\lambda\\boldsymbol{I}\\right)^{-1}\\boldsymbol{X}^T\\boldsymbol{y}=\\left(1+\\lambda\\right)^{-1}\\boldsymbol{\\beta}^{\\mathrm{OLS}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "that is the Ridge estimator scales the OLS estimator by the inverse of a factor $1+\\lambda$, and\n", + "the Ridge estimator converges to zero when the hyperparameter goes to\n", + "infinity.\n", + "\n", + "We will come back to more interpreations after we have gone through some of the statistical analysis part. \n", + "\n", + "For more discussions of Ridge and Lasso regression, [Wessel van Wieringen's](https://arxiv.org/abs/1509.09169) article is highly recommended.\n", + "Similarly, [Mehta et al's article](https://arxiv.org/abs/1803.08823) is also recommended.\n", + "\n", + "\n", + "\n", + "## A better understanding of regularization\n", + "\n", + "The parameter $\\lambda$ that we have introduced in the Ridge (and\n", + "Lasso as well) regression is often called a regularization parameter\n", + "or shrinkage parameter. It is common to call it a hyperparameter. What does it mean mathemtically?\n", + "\n", + "Here we will first look at how to analyze the difference between the\n", + "standard OLS equations and the Ridge expressions in terms of a linear\n", + "algebra analysis using the SVD algorithm. Thereafter, we will link\n", + "(see the material on the bias-variance tradeoff below) these\n", + "observation to the statisical analysis of the results. In particular\n", + "we consider how the variance of the parameters $\\boldsymbol{\\beta}$ is\n", + "affected by changing the parameter $\\lambda$.\n", + "\n", + "\n", + "We have our design matrix\n", + " $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$. With the SVD we decompose it as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X} = \\boldsymbol{U\\Sigma V^T},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{U}\\in {\\mathbb{R}}^{n\\times n}$, $\\boldsymbol{\\Sigma}\\in {\\mathbb{R}}^{n\\times p}$\n", + "and $\\boldsymbol{V}\\in {\\mathbb{R}}^{p\\times p}$.\n", + "\n", + "The matrices $\\boldsymbol{U}$ and $\\boldsymbol{V}$ are unitary/orthonormal matrices, that is in case the matrices are real we have $\\boldsymbol{U}^T\\boldsymbol{U}=\\boldsymbol{U}\\boldsymbol{U}^T=\\boldsymbol{I}$ and $\\boldsymbol{V}^T\\boldsymbol{V}=\\boldsymbol{V}\\boldsymbol{V}^T=\\boldsymbol{I}$.\n", + "\n", + "\n", + "\n", + "## Introducing the Covariance and Correlation functions\n", + "\n", + "Before we discuss the link between for example Ridge regression and the singular value decomposition, we need to remind ourselves about\n", + "the definition of the covariance and the correlation function. These are quantities \n", + "\n", + "Suppose we have defined two vectors\n", + "$\\hat{x}$ and $\\hat{y}$ with $n$ elements each. The covariance matrix $\\boldsymbol{C}$ is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x},\\boldsymbol{y}] = \\begin{bmatrix} \\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{x}] & \\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}] \\\\\n", + " \\mathrm{cov}[\\boldsymbol{y},\\boldsymbol{x}] & \\mathrm{cov}[\\boldsymbol{y},\\boldsymbol{y}] \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where for example" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}] =\\frac{1}{n} \\sum_{i=0}^{n-1}(x_i- \\overline{x})(y_i- \\overline{y}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With this definition and recalling that the variance is defined as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathrm{var}[\\boldsymbol{x}]=\\frac{1}{n} \\sum_{i=0}^{n-1}(x_i- \\overline{x})^2,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we can rewrite the covariance matrix as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x},\\boldsymbol{y}] = \\begin{bmatrix} \\mathrm{var}[\\boldsymbol{x}] & \\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}] \\\\\n", + " \\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}] & \\mathrm{var}[\\boldsymbol{y}] \\\\\n", + " \\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The covariance takes values between zero and infinity and may thus\n", + "lead to problems with loss of numerical precision for particularly\n", + "large values. It is common to scale the covariance matrix by\n", + "introducing instead the correlation matrix defined via the so-called\n", + "correlation function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathrm{corr}[\\boldsymbol{x},\\boldsymbol{y}]=\\frac{\\mathrm{cov}[\\boldsymbol{x},\\boldsymbol{y}]}{\\sqrt{\\mathrm{var}[\\boldsymbol{x}] \\mathrm{var}[\\boldsymbol{y}]}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The correlation function is then given by values $\\mathrm{corr}[\\boldsymbol{x},\\boldsymbol{y}]\n", + "\\in [-1,1]$. This avoids eventual problems with too large values. We\n", + "can then define the correlation matrix for the two vectors $\\boldsymbol{x}$\n", + "and $\\boldsymbol{y}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{K}[\\boldsymbol{x},\\boldsymbol{y}] = \\begin{bmatrix} 1 & \\mathrm{corr}[\\boldsymbol{x},\\boldsymbol{y}] \\\\\n", + " \\mathrm{corr}[\\boldsymbol{y},\\boldsymbol{x}] & 1 \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the above example this is the function we constructed using **pandas**.\n", + "\n", + "\n", + "\n", + "In our derivation of the various regression algorithms like **Ordinary Least Squares** or **Ridge regression**\n", + "we defined the design/feature matrix $\\boldsymbol{X}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\\begin{bmatrix}\n", + "x_{0,0} & x_{0,1} & x_{0,2}& \\dots & \\dots x_{0,p-1}\\\\\n", + "x_{1,0} & x_{1,1} & x_{1,2}& \\dots & \\dots x_{1,p-1}\\\\\n", + "x_{2,0} & x_{2,1} & x_{2,2}& \\dots & \\dots x_{2,p-1}\\\\\n", + "\\dots & \\dots & \\dots & \\dots \\dots & \\dots \\\\\n", + "x_{n-2,0} & x_{n-2,1} & x_{n-2,2}& \\dots & \\dots x_{n-2,p-1}\\\\\n", + "x_{n-1,0} & x_{n-1,1} & x_{n-1,2}& \\dots & \\dots x_{n-1,p-1}\\\\\n", + "\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$, with the predictors/features $p$ refering to the column numbers and the\n", + "entries $n$ being the row elements.\n", + "We can rewrite the design/feature matrix in terms of its column vectors as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\\begin{bmatrix} \\boldsymbol{x}_0 & \\boldsymbol{x}_1 & \\boldsymbol{x}_2 & \\dots & \\dots & \\boldsymbol{x}_{p-1}\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with a given vector" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_i^T = \\begin{bmatrix}x_{0,i} & x_{1,i} & x_{2,i}& \\dots & \\dots x_{n-1,i}\\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "With these definitions, we can now rewrite our $2\\times 2$\n", + "correaltion/covariance matrix in terms of a moe general design/feature\n", + "matrix $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$. This leads to a $p\\times p$\n", + "covariance matrix for the vectors $\\boldsymbol{x}_i$ with $i=0,1,\\dots,p-1$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x}] = \\begin{bmatrix}\n", + "\\mathrm{var}[\\boldsymbol{x}_0] & \\mathrm{cov}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] & \\mathrm{cov}[\\boldsymbol{x}_0,\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{cov}[\\boldsymbol{x}_0,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\mathrm{cov}[\\boldsymbol{x}_1,\\boldsymbol{x}_0] & \\mathrm{var}[\\boldsymbol{x}_1] & \\mathrm{cov}[\\boldsymbol{x}_1,\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{cov}[\\boldsymbol{x}_1,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\mathrm{cov}[\\boldsymbol{x}_2,\\boldsymbol{x}_0] & \\mathrm{cov}[\\boldsymbol{x}_2,\\boldsymbol{x}_1] & \\mathrm{var}[\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{cov}[\\boldsymbol{x}_2,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "\\mathrm{cov}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_0] & \\mathrm{cov}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_1] & \\mathrm{cov}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_{2}] & \\dots & \\dots & \\mathrm{var}[\\boldsymbol{x}_{p-1}]\\\\\n", + "\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and the correlation matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{K}[\\boldsymbol{x}] = \\begin{bmatrix}\n", + "1 & \\mathrm{corr}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] & \\mathrm{corr}[\\boldsymbol{x}_0,\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{corr}[\\boldsymbol{x}_0,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\mathrm{corr}[\\boldsymbol{x}_1,\\boldsymbol{x}_0] & 1 & \\mathrm{corr}[\\boldsymbol{x}_1,\\boldsymbol{x}_2] & \\dots & \\dots & \\mathrm{corr}[\\boldsymbol{x}_1,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\mathrm{corr}[\\boldsymbol{x}_2,\\boldsymbol{x}_0] & \\mathrm{corr}[\\boldsymbol{x}_2,\\boldsymbol{x}_1] & 1 & \\dots & \\dots & \\mathrm{corr}[\\boldsymbol{x}_2,\\boldsymbol{x}_{p-1}]\\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "\\dots & \\dots & \\dots & \\dots & \\dots & \\dots \\\\\n", + "\\mathrm{corr}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_0] & \\mathrm{corr}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_1] & \\mathrm{corr}[\\boldsymbol{x}_{p-1},\\boldsymbol{x}_{2}] & \\dots & \\dots & 1\\\\\n", + "\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The Numpy function **np.cov** calculates the covariance elements using\n", + "the factor $1/(n-1)$ instead of $1/n$ since it assumes we do not have\n", + "the exact mean values. The following simple function uses the\n", + "**np.vstack** function which takes each vector of dimension $1\\times n$\n", + "and produces a $2\\times n$ matrix $\\boldsymbol{W}$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{W} = \\begin{bmatrix} x_0 & y_0 \\\\\n", + " x_1 & y_1 \\\\\n", + " x_2 & y_2\\\\\n", + " \\dots & \\dots \\\\\n", + " x_{n-2} & y_{n-2}\\\\\n", + " x_{n-1} & y_{n-1} & \n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which in turn is converted into into the $2\\times 2$ covariance matrix\n", + "$\\boldsymbol{C}$ via the Numpy function **np.cov()**. We note that we can also calculate\n", + "the mean value of each set of samples $\\boldsymbol{x}$ etc using the Numpy\n", + "function **np.mean(x)**. We can also extract the eigenvalues of the\n", + "covariance matrix through the **np.linalg.eig()** function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "import numpy as np\n", + "n = 100\n", + "x = np.random.normal(size=n)\n", + "print(np.mean(x))\n", + "y = 4+3*x+np.random.normal(size=n)\n", + "print(np.mean(y))\n", + "W = np.vstack((x, y))\n", + "C = np.cov(W)\n", + "print(C)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The previous example can be converted into the correlation matrix by\n", + "simply scaling the matrix elements with the variances. We should also\n", + "subtract the mean values for each column. This leads to the following\n", + "code which sets up the correlations matrix for the previous example in\n", + "a more brute force way. Here we scale the mean values for each column of the design matrix, calculate the relevant mean values and variances and then finally set up the $2\\times 2$ correlation matrix (since we have only two vectors)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "n = 100\n", + "# define two vectors \n", + "x = np.random.random(size=n)\n", + "y = 4+3*x+np.random.normal(size=n)\n", + "#scaling the x and y vectors \n", + "x = x - np.mean(x)\n", + "y = y - np.mean(y)\n", + "variance_x = np.sum(x@x)/n\n", + "variance_y = np.sum(y@y)/n\n", + "print(variance_x)\n", + "print(variance_y)\n", + "cov_xy = np.sum(x@y)/n\n", + "cov_xx = np.sum(x@x)/n\n", + "cov_yy = np.sum(y@y)/n\n", + "C = np.zeros((2,2))\n", + "C[0,0]= cov_xx/variance_x\n", + "C[1,1]= cov_yy/variance_y\n", + "C[0,1]= cov_xy/np.sqrt(variance_y*variance_x)\n", + "C[1,0]= C[0,1]\n", + "print(C)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We see that the matrix elements along the diagonal are one as they\n", + "should be and that the matrix is symmetric. Furthermore, diagonalizing\n", + "this matrix we easily see that it is a positive definite matrix.\n", + "\n", + "The above procedure with **numpy** can be made more compact if we use **pandas**.\n", + "\n", + "\n", + "We whow here how we can set up the correlation matrix using **pandas**, as done in this simple code" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import pandas as pd\n", + "n = 10\n", + "x = np.random.normal(size=n)\n", + "x = x - np.mean(x)\n", + "y = 4+3*x+np.random.normal(size=n)\n", + "y = y - np.mean(y)\n", + "X = (np.vstack((x, y))).T\n", + "print(X)\n", + "Xpd = pd.DataFrame(X)\n", + "print(Xpd)\n", + "correlation_matrix = Xpd.corr()\n", + "print(correlation_matrix)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We expand this model to the Franke function discussed above." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Common imports\n", + "import numpy as np\n", + "import pandas as pd\n", + "\n", + "\n", + "def FrankeFunction(x,y):\n", + "\tterm1 = 0.75*np.exp(-(0.25*(9*x-2)**2) - 0.25*((9*y-2)**2))\n", + "\tterm2 = 0.75*np.exp(-((9*x+1)**2)/49.0 - 0.1*(9*y+1))\n", + "\tterm3 = 0.5*np.exp(-(9*x-7)**2/4.0 - 0.25*((9*y-3)**2))\n", + "\tterm4 = -0.2*np.exp(-(9*x-4)**2 - (9*y-7)**2)\n", + "\treturn term1 + term2 + term3 + term4\n", + "\n", + "\n", + "def create_X(x, y, n ):\n", + "\tif len(x.shape) > 1:\n", + "\t\tx = np.ravel(x)\n", + "\t\ty = np.ravel(y)\n", + "\n", + "\tN = len(x)\n", + "\tl = int((n+1)*(n+2)/2)\t\t# Number of elements in beta\n", + "\tX = np.ones((N,l))\n", + "\n", + "\tfor i in range(1,n+1):\n", + "\t\tq = int((i)*(i+1)/2)\n", + "\t\tfor k in range(i+1):\n", + "\t\t\tX[:,q+k] = (x**(i-k))*(y**k)\n", + "\n", + "\treturn X\n", + "\n", + "\n", + "# Making meshgrid of datapoints and compute Franke's function\n", + "n = 4\n", + "N = 100\n", + "x = np.sort(np.random.uniform(0, 1, N))\n", + "y = np.sort(np.random.uniform(0, 1, N))\n", + "z = FrankeFunction(x, y)\n", + "X = create_X(x, y, n=n) \n", + "\n", + "Xpd = pd.DataFrame(X)\n", + "# subtract the mean values and set up the covariance matrix\n", + "Xpd = Xpd - Xpd.mean()\n", + "covariance_matrix = Xpd.cov()\n", + "print(covariance_matrix)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We note here that the covariance is zero for the first rows and\n", + "columns since all matrix elements in the design matrix were set to one\n", + "(we are fitting the function in terms of a polynomial of degree $n$).\n", + "\n", + "This means that the variance for these elements will be zero and will\n", + "cause problems when we set up the correlation matrix. We can simply\n", + "drop these elements and construct a correlation\n", + "matrix without these elements. \n", + "\n", + "\n", + "\n", + "\n", + "We can rewrite the covariance matrix in a more compact form in terms of the design/feature matrix $\\boldsymbol{X}$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x}] = \\frac{1}{n}\\boldsymbol{X}^T\\boldsymbol{X}= \\mathbb{E}[\\boldsymbol{X}^T\\boldsymbol{X}].\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "To see this let us simply look at a design matrix $\\boldsymbol{X}\\in {\\mathbb{R}}^{2\\times 2}$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{X}=\\begin{bmatrix}\n", + "x_{00} & x_{01}\\\\\n", + "x_{10} & x_{11}\\\\\n", + "\\end{bmatrix}=\\begin{bmatrix}\n", + "\\boldsymbol{x}_{0} & \\boldsymbol{x}_{1}\\\\\n", + "\\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we then compute the expectation value" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbb{E}[\\boldsymbol{X}^T\\boldsymbol{X}] = \\frac{1}{n}\\boldsymbol{X}^T\\boldsymbol{X}=\\begin{bmatrix}\n", + "x_{00}^2+x_{01}^2 & x_{00}x_{10}+x_{01}x_{11}\\\\\n", + "x_{10}x_{00}+x_{11}x_{01} & x_{10}^2+x_{11}^2\\\\\n", + "\\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which is just" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{C}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] = \\boldsymbol{C}[\\boldsymbol{x}]=\\begin{bmatrix} \\mathrm{var}[\\boldsymbol{x}_0] & \\mathrm{cov}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] \\\\\n", + " \\mathrm{cov}[\\boldsymbol{x}_1,\\boldsymbol{x}_0] & \\mathrm{var}[\\boldsymbol{x}_1] \\\\\n", + " \\end{bmatrix},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we wrote $$\\boldsymbol{C}[\\boldsymbol{x}_0,\\boldsymbol{x}_1] = \\boldsymbol{C}[\\boldsymbol{x}]$$ to indicate that this the covariance of the vectors $\\boldsymbol{x}$ of the design/feature matrix $\\boldsymbol{X}$.\n", + "\n", + "It is easy to generalize this to a matrix $\\boldsymbol{X}\\in {\\mathbb{R}}^{n\\times p}$.\n", + "\n", + "\n", + "## Linking with SVD" + ] + } + ], + "metadata": {}, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/doc/LectureNotes/chapter4.ipynb b/doc/LectureNotes/chapter4.ipynb new file mode 100644 index 000000000..80fb450d4 --- /dev/null +++ b/doc/LectureNotes/chapter4.ipynb @@ -0,0 +1,2875 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Logistic Regression\n", + "\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK3155/h20/forelesningsvideoer/LectureSeptember18.mp4?vrtx=view-as-webpage)\n", + "\n", + "\n", + "## Logistic Regression\n", + "\n", + "In linear regression our main interest was centered on learning the\n", + "coefficients of a functional fit (say a polynomial) in order to be\n", + "able to predict the response of a continuous variable on some unseen\n", + "data. The fit to the continuous variable $y_i$ is based on some\n", + "independent variables $\\hat{x}_i$. Linear regression resulted in\n", + "analytical expressions for standard ordinary Least Squares or Ridge\n", + "regression (in terms of matrices to invert) for several quantities,\n", + "ranging from the variance and thereby the confidence intervals of the\n", + "parameters $\\hat{\\beta}$ to the mean squared error. If we can invert\n", + "the product of the design matrices, linear regression gives then a\n", + "simple recipe for fitting our data.\n", + "\n", + "\n", + "Classification problems, however, are concerned with outcomes taking\n", + "the form of discrete variables (i.e. categories). We may for example,\n", + "on the basis of DNA sequencing for a number of patients, like to find\n", + "out which mutations are important for a certain disease; or based on\n", + "scans of various patients' brains, figure out if there is a tumor or\n", + "not; or given a specific physical system, we'd like to identify its\n", + "state, say whether it is an ordered or disordered system (typical\n", + "situation in solid state physics); or classify the status of a\n", + "patient, whether she/he has a stroke or not and many other similar\n", + "situations.\n", + "\n", + "The most common situation we encounter when we apply logistic\n", + "regression is that of two possible outcomes, normally denoted as a\n", + "binary outcome, true or false, positive or negative, success or\n", + "failure etc.\n", + "\n", + "\n", + "Logistic regression will also serve as our stepping stone towards\n", + "neural network algorithms and supervised deep learning. For logistic\n", + "learning, the minimization of the cost function leads to a non-linear\n", + "equation in the parameters $\\hat{\\beta}$. The optimization of the\n", + "problem calls therefore for minimization algorithms. This forms the\n", + "bottle neck of all machine learning algorithms, namely how to find\n", + "reliable minima of a multi-variable function. This leads us to the\n", + "family of gradient descent methods. The latter are the working horses\n", + "of basically all modern machine learning algorithms.\n", + "\n", + "We note also that many of the topics discussed here on logistic \n", + "regression are also commonly used in modern supervised Deep Learning\n", + "models, as we will see later.\n", + "\n", + "\n", + "\n", + "## Basics\n", + "\n", + "We consider the case where the dependent variables, also called the\n", + "responses or the outcomes, $y_i$ are discrete and only take values\n", + "from $k=0,\\dots,K-1$ (i.e. $K$ classes).\n", + "\n", + "The goal is to predict the\n", + "output classes from the design matrix $\\hat{X}\\in\\mathbb{R}^{n\\times p}$\n", + "made of $n$ samples, each of which carries $p$ features or predictors. The\n", + "primary goal is to identify the classes to which new unseen samples\n", + "belong.\n", + "\n", + "Let us specialize to the case of two classes only, with outputs\n", + "$y_i=0$ and $y_i=1$. Our outcomes could represent the status of a\n", + "credit card user that could default or not on her/his credit card\n", + "debt. That is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "y_i = \\begin{bmatrix} 0 & \\mathrm{no}\\\\ 1 & \\mathrm{yes} \\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Before moving to the logistic model, let us try to use our linear\n", + "regression model to classify these two outcomes. We could for example\n", + "fit a linear model to the default case if $y_i > 0.5$ and the no\n", + "default case $y_i \\leq 0.5$.\n", + "\n", + "We would then have our \n", + "weighted linear combination, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "
\n", + "\n", + "$$\n", + "\\begin{equation}\n", + "\\hat{y} = \\hat{X}^T\\hat{\\beta} + \\hat{\\epsilon},\n", + "\\label{_auto1} \\tag{1}\n", + "\\end{equation}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\hat{y}$ is a vector representing the possible outcomes, $\\hat{X}$ is our\n", + "$n\\times p$ design matrix and $\\hat{\\beta}$ represents our estimators/predictors.\n", + "\n", + "\n", + "The main problem with our function is that it takes values on the\n", + "entire real axis. In the case of logistic regression, however, the\n", + "labels $y_i$ are discrete variables. A typical example is the credit\n", + "card data discussed below here, where we can set the state of\n", + "defaulting the debt to $y_i=1$ and not to $y_i=0$ for one the persons\n", + "in the data set (see the full example below).\n", + "\n", + "One simple way to get a discrete output is to have sign\n", + "functions that map the output of a linear regressor to values $\\{0,1\\}$,\n", + "$f(s_i)=sign(s_i)=1$ if $s_i\\ge 0$ and 0 if otherwise. \n", + "We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron\" model in the machine learning\n", + "literature. This model is extremely simple. However, in many cases it is more\n", + "favorable to use a ``soft\" classifier that outputs\n", + "the probability of a given category. This leads us to the logistic function.\n", + "\n", + "\n", + "The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "%matplotlib inline\n", + "\n", + "# Common imports\n", + "import os\n", + "import numpy as np\n", + "import pandas as pd\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import LinearRegression, Ridge, Lasso\n", + "from sklearn.model_selection import train_test_split\n", + "from sklearn.utils import resample\n", + "from sklearn.metrics import mean_squared_error\n", + "from IPython.display import display\n", + "from pylab import plt, mpl\n", + "plt.style.use('seaborn')\n", + "mpl.rcParams['font.family'] = 'serif'\n", + "\n", + "# Where to save the figures and data files\n", + "PROJECT_ROOT_DIR = \"Results\"\n", + "FIGURE_ID = \"Results/FigureFiles\"\n", + "DATA_ID = \"DataFiles/\"\n", + "\n", + "if not os.path.exists(PROJECT_ROOT_DIR):\n", + " os.mkdir(PROJECT_ROOT_DIR)\n", + "\n", + "if not os.path.exists(FIGURE_ID):\n", + " os.makedirs(FIGURE_ID)\n", + "\n", + "if not os.path.exists(DATA_ID):\n", + " os.makedirs(DATA_ID)\n", + "\n", + "def image_path(fig_id):\n", + " return os.path.join(FIGURE_ID, fig_id)\n", + "\n", + "def data_path(dat_id):\n", + " return os.path.join(DATA_ID, dat_id)\n", + "\n", + "def save_fig(fig_id):\n", + " plt.savefig(image_path(fig_id) + \".png\", format='png')\n", + "\n", + "infile = open(data_path(\"chddata.csv\"),'r')\n", + "\n", + "# Read the chd data as csv file and organize the data into arrays with age group, age, and chd\n", + "chd = pd.read_csv(infile, names=('ID', 'Age', 'Agegroup', 'CHD'))\n", + "chd.columns = ['ID', 'Age', 'Agegroup', 'CHD']\n", + "output = chd['CHD']\n", + "age = chd['Age']\n", + "agegroup = chd['Agegroup']\n", + "numberID = chd['ID'] \n", + "display(chd)\n", + "\n", + "plt.scatter(age, output, marker='o')\n", + "plt.axis([18,70.0,-0.1, 1.2])\n", + "plt.xlabel(r'Age')\n", + "plt.ylabel(r'CHD')\n", + "plt.title(r'Age distribution and Coronary heart disease')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "What we could attempt however is to plot the mean value for each group." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "agegroupmean = np.array([0.1, 0.133, 0.250, 0.333, 0.462, 0.625, 0.765, 0.800])\n", + "group = np.array([1, 2, 3, 4, 5, 6, 7, 8])\n", + "plt.plot(group, agegroupmean, \"r-\")\n", + "plt.axis([0,9,0, 1.0])\n", + "plt.xlabel(r'Age group')\n", + "plt.ylabel(r'CHD mean values')\n", + "plt.title(r'Mean values for each age group')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We are now trying to find a function $f(y\\vert x)$, that is a function which gives us an expected value for the output $y$ with a given input $x$.\n", + "In standard linear regression with a linear dependence on $x$, we would write this in terms of our model" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(y_i\\vert x_i)=\\beta_0+\\beta_1 x_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This expression implies however that $f(y_i\\vert x_i)$ could take any\n", + "value from minus infinity to plus infinity. If we however let\n", + "$f(y\\vert y)$ be represented by the mean value, the above example\n", + "shows us that we can constrain the function to take values between\n", + "zero and one, that is we have $0 \\le f(y_i\\vert x_i) \\le 1$. Looking\n", + "at our last curve we see also that it has an S-shaped form. This leads\n", + "us to a very popular model for the function $f$, namely the so-called\n", + "Sigmoid function or logistic model. We will consider this function as\n", + "representing the probability for finding a value of $y_i$ with a given\n", + "$x_i$.\n", + "\n", + "\n", + "## The logistic function\n", + "\n", + "Another widely studied model, is the so-called \n", + "perceptron model, which is an example of a \"hard classification\" model. We\n", + "will encounter this model when we discuss neural networks as\n", + "well. Each datapoint is deterministically assigned to a category (i.e\n", + "$y_i=0$ or $y_i=1$). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a \"soft\"\n", + "classifier that outputs the probability of a given category rather\n", + "than a single value. For example, given $x_i$, the classifier\n", + "outputs the probability of being in a category $k$. Logistic regression\n", + "is the most common example of a so-called soft classifier. In logistic\n", + "regression, the probability that a data point $x_i$\n", + "belongs to a category $y_i=\\{0,1\\}$ is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(t) = \\frac{1}{1+\\mathrm \\exp{-t}}=\\frac{\\exp{t}}{1+\\mathrm \\exp{t}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Note that $1-p(t)= p(-t)$.\n", + "\n", + "## Examples of likelihood functions used in logistic regression and nueral networks\n", + "\n", + "\n", + "The following code plots the logistic function, the step function and other functions we will encounter from here and on." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\"\"\"The sigmoid function (or the logistic curve) is a\n", + "function that takes any real number, z, and outputs a number (0,1).\n", + "It is useful in neural networks for assigning weights on a relative scale.\n", + "The value z is the weighted sum of parameters involved in the learning algorithm.\"\"\"\n", + "\n", + "import numpy\n", + "import matplotlib.pyplot as plt\n", + "import math as mt\n", + "\n", + "z = numpy.arange(-5, 5, .1)\n", + "sigma_fn = numpy.vectorize(lambda z: 1/(1+numpy.exp(-z)))\n", + "sigma = sigma_fn(z)\n", + "\n", + "fig = plt.figure()\n", + "ax = fig.add_subplot(111)\n", + "ax.plot(z, sigma)\n", + "ax.set_ylim([-0.1, 1.1])\n", + "ax.set_xlim([-5,5])\n", + "ax.grid(True)\n", + "ax.set_xlabel('z')\n", + "ax.set_title('sigmoid function')\n", + "\n", + "plt.show()\n", + "\n", + "\"\"\"Step Function\"\"\"\n", + "z = numpy.arange(-5, 5, .02)\n", + "step_fn = numpy.vectorize(lambda z: 1.0 if z >= 0.0 else 0.0)\n", + "step = step_fn(z)\n", + "\n", + "fig = plt.figure()\n", + "ax = fig.add_subplot(111)\n", + "ax.plot(z, step)\n", + "ax.set_ylim([-0.5, 1.5])\n", + "ax.set_xlim([-5,5])\n", + "ax.grid(True)\n", + "ax.set_xlabel('z')\n", + "ax.set_title('step function')\n", + "\n", + "plt.show()\n", + "\n", + "\"\"\"tanh Function\"\"\"\n", + "z = numpy.arange(-2*mt.pi, 2*mt.pi, 0.1)\n", + "t = numpy.tanh(z)\n", + "\n", + "fig = plt.figure()\n", + "ax = fig.add_subplot(111)\n", + "ax.plot(z, t)\n", + "ax.set_ylim([-1.0, 1.0])\n", + "ax.set_xlim([-2*mt.pi,2*mt.pi])\n", + "ax.grid(True)\n", + "ax.set_xlabel('z')\n", + "ax.set_title('tanh function')\n", + "\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We assume now that we have two classes with $y_i$ either $0$ or $1$. Furthermore we assume also that we have only two parameters $\\beta$ in our fitting of the Sigmoid function, that is we define probabilities" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "p(y_i=1|x_i,\\hat{\\beta}) &= \\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}},\\nonumber\\\\\n", + "p(y_i=0|x_i,\\hat{\\beta}) &= 1 - p(y_i=1|x_i,\\hat{\\beta}),\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\hat{\\beta}$ are the weights we wish to extract from data, in our case $\\beta_0$ and $\\beta_1$. \n", + "\n", + "Note that we used" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(y_i=0\\vert x_i, \\hat{\\beta}) = 1-p(y_i=1\\vert x_i, \\hat{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In order to define the total likelihood for all possible outcomes from a \n", + "dataset $\\mathcal{D}=\\{(y_i,x_i)\\}$, with the binary labels\n", + "$y_i\\in\\{0,1\\}$ and where the data points are drawn independently, we use the so-called [Maximum Likelihood Estimation](https://en.wikipedia.org/wiki/Maximum_likelihood_estimation) (MLE) principle. \n", + "We aim thus at maximizing \n", + "the probability of seeing the observed data. We can then approximate the \n", + "likelihood in terms of the product of the individual probabilities of a specific outcome $y_i$, that is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "P(\\mathcal{D}|\\hat{\\beta})& = \\prod_{i=1}^n \\left[p(y_i=1|x_i,\\hat{\\beta})\\right]^{y_i}\\left[1-p(y_i=1|x_i,\\hat{\\beta}))\\right]^{1-y_i}\\nonumber \\\\\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "from which we obtain the log-likelihood and our **cost/loss** function" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathcal{C}(\\hat{\\beta}) = \\sum_{i=1}^n \\left( y_i\\log{p(y_i=1|x_i,\\hat{\\beta})} + (1-y_i)\\log\\left[1-p(y_i=1|x_i,\\hat{\\beta}))\\right]\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Reordering the logarithms, we can rewrite the **cost/loss** function as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathcal{C}(\\hat{\\beta}) = \\sum_{i=1}^n \\left(y_i(\\beta_0+\\beta_1x_i) -\\log{(1+\\exp{(\\beta_0+\\beta_1x_i)})}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to $\\beta$.\n", + "Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathcal{C}(\\hat{\\beta})=-\\sum_{i=1}^n \\left(y_i(\\beta_0+\\beta_1x_i) -\\log{(1+\\exp{(\\beta_0+\\beta_1x_i)})}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This equation is known in statistics as the **cross entropy**. Finally, we note that just as in linear regression, \n", + "in practice we often supplement the cross-entropy with additional regularization terms, usually $L_1$ and $L_2$ regularization as we did for Ridge and Lasso regression.\n", + "\n", + "\n", + "The cross entropy is a convex function of the weights $\\hat{\\beta}$ and,\n", + "therefore, any local minimizer is a global minimizer. \n", + "\n", + "\n", + "Minimizing this\n", + "cost function with respect to the two parameters $\\beta_0$ and $\\beta_1$ we obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\hat{\\beta})}{\\partial \\beta_0} = -\\sum_{i=1}^n \\left(y_i -\\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}}\\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\hat{\\beta})}{\\partial \\beta_1} = -\\sum_{i=1}^n \\left(y_ix_i -x_i\\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Let us now define a vector $\\hat{y}$ with $n$ elements $y_i$, an\n", + "$n\\times p$ matrix $\\hat{X}$ which contains the $x_i$ values and a\n", + "vector $\\hat{p}$ of fitted probabilities $p(y_i\\vert x_i,\\hat{\\beta})$. We can rewrite in a more compact form the first\n", + "derivative of cost function as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\hat{\\beta})}{\\partial \\hat{\\beta}} = -\\hat{X}^T\\left(\\hat{y}-\\hat{p}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we in addition define a diagonal matrix $\\hat{W}$ with elements \n", + "$p(y_i\\vert x_i,\\hat{\\beta})(1-p(y_i\\vert x_i,\\hat{\\beta})$, we can obtain a compact expression of the second derivative as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial^2 \\mathcal{C}(\\hat{\\beta})}{\\partial \\hat{\\beta}\\partial \\hat{\\beta}^T} = \\hat{X}^T\\hat{W}\\hat{X}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with $p$ predictors" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{ \\frac{p(\\hat{\\beta}\\hat{x})}{1-p(\\hat{\\beta}\\hat{x})}} = \\beta_0+\\beta_1x_1+\\beta_2x_2+\\dots+\\beta_px_p.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Here we defined $\\hat{x}=[1,x_1,x_2,\\dots,x_p]$ and $\\hat{\\beta}=[\\beta_0, \\beta_1, \\dots, \\beta_p]$ leading to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(\\hat{\\beta}\\hat{x})=\\frac{ \\exp{(\\beta_0+\\beta_1x_1+\\beta_2x_2+\\dots+\\beta_px_p)}}{1+\\exp{(\\beta_0+\\beta_1x_1+\\beta_2x_2+\\dots+\\beta_px_p)}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Till now we have mainly focused on two classes, the so-called binary\n", + "system. Suppose we wish to extend to $K$ classes. Let us for the sake\n", + "of simplicity assume we have only two predictors. We have then following model" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{\\frac{p(C=1\\vert x)}{p(K\\vert x)}} = \\beta_{10}+\\beta_{11}x_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{\\frac{p(C=2\\vert x)}{p(K\\vert x)}} = \\beta_{20}+\\beta_{21}x_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and so on till the class $C=K-1$ class" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\log{\\frac{p(C=K-1\\vert x)}{p(K\\vert x)}} = \\beta_{(K-1)0}+\\beta_{(K-1)1}x_1,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and the model is specified in term of $K-1$ so-called log-odds or\n", + "**logit** transformations.\n", + "\n", + "\n", + "\n", + "In our discussion of neural networks we will encounter the above again\n", + "in terms of a slightly modified function, the so-called **Softmax** function.\n", + "\n", + "The softmax function is used in various multiclass classification\n", + "methods, such as multinomial logistic regression (also known as\n", + "softmax regression), multiclass linear discriminant analysis, naive\n", + "Bayes classifiers, and artificial neural networks. Specifically, in\n", + "multinomial logistic regression and linear discriminant analysis, the\n", + "input to the function is the result of $K$ distinct linear functions,\n", + "and the predicted probability for the $k$-th class given a sample\n", + "vector $\\hat{x}$ and a weighting vector $\\hat{\\beta}$ is (with two\n", + "predictors):" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(C=k\\vert \\mathbf {x} )=\\frac{\\exp{(\\beta_{k0}+\\beta_{k1}x_1)}}{1+\\sum_{l=1}^{K-1}\\exp{(\\beta_{l0}+\\beta_{l1}x_1)}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "It is easy to extend to more predictors. The final class is" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "p(C=K\\vert \\mathbf {x} )=\\frac{1}{1+\\sum_{l=1}^{K-1}\\exp{(\\beta_{l0}+\\beta_{l1}x_1)}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and they sum to one. Our earlier discussions were all specialized to\n", + "the case with two classes only. It is easy to see from the above that\n", + "what we derived earlier is compatible with these equations.\n", + "\n", + "To find the optimal parameters we would typically use a gradient\n", + "descent method. Newton's method and gradient descent methods are\n", + "discussed in the material on [optimization\n", + "methods](https://compphysics.github.io/MachineLearning/doc/pub/Splines/html/Splines-bs.html).\n", + "\n", + "## Wisconsin Cancer Data\n", + "\n", + "We show here how we can use a simple regression case on the breast\n", + "cancer data using Logistic regression as our algorithm for\n", + "classification." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.linear_model import LogisticRegression\n", + "\n", + "# Load the data\n", + "cancer = load_breast_cancer()\n", + "\n", + "X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)\n", + "print(X_train.shape)\n", + "print(X_test.shape)\n", + "# Logistic Regression\n", + "logreg = LogisticRegression(solver='lbfgs')\n", + "logreg.fit(X_train, y_train)\n", + "print(\"Test set accuracy with Logistic Regression: {:.2f}\".format(logreg.score(X_test,y_test)))\n", + "#now scale the data\n", + "from sklearn.preprocessing import StandardScaler\n", + "scaler = StandardScaler()\n", + "scaler.fit(X_train)\n", + "X_train_scaled = scaler.transform(X_train)\n", + "X_test_scaled = scaler.transform(X_test)\n", + "# Logistic Regression\n", + "logreg.fit(X_train_scaled, y_train)\n", + "print(\"Test set accuracy Logistic Regression with scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In addition to the above scores, we could also study the covariance (and the correlation matrix).\n", + "We use **Pandas** to compute the correlation matrix." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.linear_model import LogisticRegression\n", + "cancer = load_breast_cancer()\n", + "import pandas as pd\n", + "# Making a data frame\n", + "cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)\n", + "\n", + "fig, axes = plt.subplots(15,2,figsize=(10,20))\n", + "malignant = cancer.data[cancer.target == 0]\n", + "benign = cancer.data[cancer.target == 1]\n", + "ax = axes.ravel()\n", + "\n", + "for i in range(30):\n", + " _, bins = np.histogram(cancer.data[:,i], bins =50)\n", + " ax[i].hist(malignant[:,i], bins = bins, alpha = 0.5)\n", + " ax[i].hist(benign[:,i], bins = bins, alpha = 0.5)\n", + " ax[i].set_title(cancer.feature_names[i])\n", + " ax[i].set_yticks(())\n", + "ax[0].set_xlabel(\"Feature magnitude\")\n", + "ax[0].set_ylabel(\"Frequency\")\n", + "ax[0].legend([\"Malignant\", \"Benign\"], loc =\"best\")\n", + "fig.tight_layout()\n", + "plt.show()\n", + "\n", + "import seaborn as sns\n", + "correlation_matrix = cancerpd.corr().round(1)\n", + "# use the heatmap function from seaborn to plot the correlation matrix\n", + "# annot = True to print the values inside the square\n", + "plt.figure(figsize=(15,8))\n", + "sns.heatmap(data=correlation_matrix, annot=True)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the above example we note two things. In the first plot we display\n", + "the overlap of benign and malignant tumors as functions of the various\n", + "features in the Wisconsing breast cancer data set. We see that for\n", + "some of the features we can distinguish clearly the benign and\n", + "malignant cases while for other features we cannot. This can point to\n", + "us which features may be of greater interest when we wish to classify\n", + "a benign or not benign tumour.\n", + "\n", + "In the second figure we have computed the so-called correlation\n", + "matrix, which in our case with thirty features becomes a $30\\times 30$\n", + "matrix.\n", + "\n", + "We constructed this matrix using **pandas** via the statements" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "cancerpd = pd.DataFrame(cancer.data, columns=cancer.feature_names)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and then" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "correlation_matrix = cancerpd.corr().round(1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Diagonalizing this matrix we can in turn say something about which\n", + "features are of relevance and which are not. This leads us to\n", + "the classical Principal Component Analysis (PCA) theorem with\n", + "applications. This will be discussed later this semester ([week 43](https://compphysics.github.io/MachineLearning/doc/pub/week43/html/week43-bs.html))." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "from sklearn.model_selection import train_test_split \n", + "from sklearn.datasets import load_breast_cancer\n", + "from sklearn.linear_model import LogisticRegression\n", + "\n", + "# Load the data\n", + "cancer = load_breast_cancer()\n", + "\n", + "X_train, X_test, y_train, y_test = train_test_split(cancer.data,cancer.target,random_state=0)\n", + "print(X_train.shape)\n", + "print(X_test.shape)\n", + "# Logistic Regression\n", + "logreg = LogisticRegression(solver='lbfgs')\n", + "logreg.fit(X_train, y_train)\n", + "print(\"Test set accuracy with Logistic Regression: {:.2f}\".format(logreg.score(X_test,y_test)))\n", + "#now scale the data\n", + "from sklearn.preprocessing import StandardScaler\n", + "scaler = StandardScaler()\n", + "scaler.fit(X_train)\n", + "X_train_scaled = scaler.transform(X_train)\n", + "X_test_scaled = scaler.transform(X_test)\n", + "# Logistic Regression\n", + "logreg.fit(X_train_scaled, y_train)\n", + "print(\"Test set accuracy Logistic Regression with scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))\n", + "\n", + "\n", + "from sklearn.preprocessing import LabelEncoder\n", + "from sklearn.model_selection import cross_validate\n", + "#Cross validation\n", + "accuracy = cross_validate(logreg,X_test_scaled,y_test,cv=10)['test_score']\n", + "print(accuracy)\n", + "print(\"Test set accuracy with Logistic Regression and scaled data: {:.2f}\".format(logreg.score(X_test_scaled,y_test)))\n", + "\n", + "\n", + "import scikitplot as skplt\n", + "y_pred = logreg.predict(X_test_scaled)\n", + "skplt.metrics.plot_confusion_matrix(y_test, y_pred, normalize=True)\n", + "plt.show()\n", + "y_probas = logreg.predict_proba(X_test_scaled)\n", + "skplt.metrics.plot_roc(y_test, y_probas)\n", + "plt.show()\n", + "skplt.metrics.plot_cumulative_gain(y_test, y_probas)\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Optimization, the central part of any Machine Learning algortithm\n", + "\n", + "Almost every problem in machine learning and data science starts with\n", + "a dataset $X$, a model $g(\\beta)$, which is a function of the\n", + "parameters $\\beta$ and a cost function $C(X, g(\\beta))$ that allows\n", + "us to judge how well the model $g(\\beta)$ explains the observations\n", + "$X$. The model is fit by finding the values of $\\beta$ that minimize\n", + "the cost function. Ideally we would be able to solve for $\\beta$\n", + "analytically, however this is not possible in general and we must use\n", + "some approximative/numerical method to compute the minimum.\n", + "\n", + "\n", + "\n", + "## Revisiting our Logistic Regression case\n", + "\n", + "In our discussion on Logistic Regression we studied the \n", + "case of\n", + "two classes, with $y_i$ either\n", + "$0$ or $1$. Furthermore we assumed also that we have only two\n", + "parameters $\\beta$ in our fitting, that is we\n", + "defined probabilities" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{align*}\n", + "p(y_i=1|x_i,\\boldsymbol{\\beta}) &= \\frac{\\exp{(\\beta_0+\\beta_1x_i)}}{1+\\exp{(\\beta_0+\\beta_1x_i)}},\\nonumber\\\\\n", + "p(y_i=0|x_i,\\boldsymbol{\\beta}) &= 1 - p(y_i=1|x_i,\\boldsymbol{\\beta}),\n", + "\\end{align*}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\boldsymbol{\\beta}$ are the weights we wish to extract from data, in our case $\\beta_0$ and $\\beta_1$. \n", + "\n", + "\n", + "## The equations to solve\n", + "\n", + "Our compact equations used a definition of a vector $\\boldsymbol{y}$ with $n$\n", + "elements $y_i$, an $n\\times p$ matrix $\\boldsymbol{X}$ which contains the\n", + "$x_i$ values and a vector $\\boldsymbol{p}$ of fitted probabilities\n", + "$p(y_i\\vert x_i,\\boldsymbol{\\beta})$. We rewrote in a more compact form\n", + "the first derivative of the cost function as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}} = -\\boldsymbol{X}^T\\left(\\boldsymbol{y}-\\boldsymbol{p}\\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "If we in addition define a diagonal matrix $\\boldsymbol{W}$ with elements \n", + "$p(y_i\\vert x_i,\\boldsymbol{\\beta})(1-p(y_i\\vert x_i,\\boldsymbol{\\beta})$, we can obtain a compact expression of the second derivative as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\frac{\\partial^2 \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}\\partial \\boldsymbol{\\beta}^T} = \\boldsymbol{X}^T\\boldsymbol{W}\\boldsymbol{X}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This defines what is called the Hessian matrix.\n", + "\n", + "\n", + "## Solving using Newton-Raphson's method\n", + "\n", + "If we can set up these equations, Newton-Raphson's iterative method is normally the method of choice. It requires however that we can compute in an efficient way the matrices that define the first and second derivatives. \n", + "\n", + "Our iterative scheme is then given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{new}} = \\boldsymbol{\\beta}^{\\mathrm{old}}-\\left(\\frac{\\partial^2 \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}\\partial \\boldsymbol{\\beta}^T}\\right)^{-1}_{\\boldsymbol{\\beta}^{\\mathrm{old}}}\\times \\left(\\frac{\\partial \\mathcal{C}(\\boldsymbol{\\beta})}{\\partial \\boldsymbol{\\beta}}\\right)_{\\boldsymbol{\\beta}^{\\mathrm{old}}},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or in matrix form as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{\\beta}^{\\mathrm{new}} = \\boldsymbol{\\beta}^{\\mathrm{old}}-\\left(\\boldsymbol{X}^T\\boldsymbol{W}\\boldsymbol{X} \\right)^{-1}\\times \\left(-\\boldsymbol{X}^T(\\boldsymbol{y}-\\boldsymbol{p}) \\right)_{\\boldsymbol{\\beta}^{\\mathrm{old}}}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The right-hand side is computed with the old values of $\\beta$. \n", + "\n", + "If we can compute these matrices, in particular the Hessian, the above is often the easiest method to implement. \n", + "\n", + "\n", + "\n", + "## Brief reminder on Newton-Raphson's method\n", + "\n", + "Let us quickly remind ourselves how we derive the above method.\n", + "\n", + "Perhaps the most celebrated of all one-dimensional root-finding\n", + "routines is Newton's method, also called the Newton-Raphson\n", + "method. This method requires the evaluation of both the\n", + "function $f$ and its derivative $f'$ at arbitrary points. \n", + "If you can only calculate the derivative\n", + "numerically and/or your function is not of the smooth type, we\n", + "normally discourage the use of this method.\n", + "\n", + "\n", + "## The equations\n", + "\n", + "The Newton-Raphson formula consists geometrically of extending the\n", + "tangent line at a current point until it crosses zero, then setting\n", + "the next guess to the abscissa of that zero-crossing. The mathematics\n", + "behind this method is rather simple. Employing a Taylor expansion for\n", + "$x$ sufficiently close to the solution $s$, we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "
\n", + "\n", + "$$\n", + "f(s)=0=f(x)+(s-x)f'(x)+\\frac{(s-x)^2}{2}f''(x) +\\dots.\n", + " \\label{eq:taylornr} \\tag{2}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "For small enough values of the function and for well-behaved\n", + "functions, the terms beyond linear are unimportant, hence we obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(x)+(s-x)f'(x)\\approx 0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "yielding" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "s\\approx x-\\frac{f(x)}{f'(x)}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Having in mind an iterative procedure, it is natural to start iterating with" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "x_{n+1}=x_n-\\frac{f(x_n)}{f'(x_n)}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Simple geometric interpretation\n", + "\n", + "The above is Newton-Raphson's method. It has a simple geometric\n", + "interpretation, namely $x_{n+1}$ is the point where the tangent from\n", + "$(x_n,f(x_n))$ crosses the $x$-axis. Close to the solution,\n", + "Newton-Raphson converges fast to the desired result. However, if we\n", + "are far from a root, where the higher-order terms in the series are\n", + "important, the Newton-Raphson formula can give grossly inaccurate\n", + "results. For instance, the initial guess for the root might be so far\n", + "from the true root as to let the search interval include a local\n", + "maximum or minimum of the function. If an iteration places a trial\n", + "guess near such a local extremum, so that the first derivative nearly\n", + "vanishes, then Newton-Raphson may fail totally\n", + "\n", + "\n", + "\n", + "## Extending to more than one variable\n", + "\n", + "Newton's method can be generalized to systems of several non-linear equations\n", + "and variables. Consider the case with two equations" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{array}{cc} f_1(x_1,x_2) &=0\\\\\n", + " f_2(x_1,x_2) &=0,\\end{array}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which we Taylor expand to obtain" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\begin{array}{cc} 0=f_1(x_1+h_1,x_2+h_2)=&f_1(x_1,x_2)+h_1\n", + " \\partial f_1/\\partial x_1+h_2\n", + " \\partial f_1/\\partial x_2+\\dots\\\\\n", + " 0=f_2(x_1+h_1,x_2+h_2)=&f_2(x_1,x_2)+h_1\n", + " \\partial f_2/\\partial x_1+h_2\n", + " \\partial f_2/\\partial x_2+\\dots\n", + " \\end{array}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Defining the Jacobian matrix ${\\bf \\boldsymbol{J}}$ we have" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "{\\bf \\boldsymbol{J}}=\\left( \\begin{array}{cc}\n", + " \\partial f_1/\\partial x_1 & \\partial f_1/\\partial x_2 \\\\\n", + " \\partial f_2/\\partial x_1 &\\partial f_2/\\partial x_2\n", + " \\end{array} \\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "we can rephrase Newton's method as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\left(\\begin{array}{c} x_1^{n+1} \\\\ x_2^{n+1} \\end{array} \\right)=\n", + "\\left(\\begin{array}{c} x_1^{n} \\\\ x_2^{n} \\end{array} \\right)+\n", + "\\left(\\begin{array}{c} h_1^{n} \\\\ h_2^{n} \\end{array} \\right),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where we have defined" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\left(\\begin{array}{c} h_1^{n} \\\\ h_2^{n} \\end{array} \\right)=\n", + " -{\\bf \\boldsymbol{J}}^{-1}\n", + " \\left(\\begin{array}{c} f_1(x_1^{n},x_2^{n}) \\\\ f_2(x_1^{n},x_2^{n}) \\end{array} \\right).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We need thus to compute the inverse of the Jacobian matrix and it\n", + "is to understand that difficulties may\n", + "arise in case ${\\bf \\boldsymbol{J}}$ is nearly singular.\n", + "\n", + "It is rather straightforward to extend the above scheme to systems of\n", + "more than two non-linear equations. In our case, the Jacobian matrix is given by the Hessian that represents the second derivative of cost function. \n", + "\n", + "\n", + "\n", + "\n", + "## Steepest descent\n", + "\n", + "The basic idea of gradient descent is\n", + "that a function $F(\\mathbf{x})$, \n", + "$\\mathbf{x} \\equiv (x_1,\\cdots,x_n)$, decreases fastest if one goes from $\\bf {x}$ in the\n", + "direction of the negative gradient $-\\nabla F(\\mathbf{x})$.\n", + "\n", + "It can be shown that if" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{x}_{k+1} = \\mathbf{x}_k - \\gamma_k \\nabla F(\\mathbf{x}_k),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $\\gamma_k > 0$.\n", + "\n", + "For $\\gamma_k$ small enough, then $F(\\mathbf{x}_{k+1}) \\leq\n", + "F(\\mathbf{x}_k)$. This means that for a sufficiently small $\\gamma_k$\n", + "we are always moving towards smaller function values, i.e a minimum.\n", + "\n", + "\n", + "## More on Steepest descent\n", + "\n", + "The previous observation is the basis of the method of steepest\n", + "descent, which is also referred to as just gradient descent (GD). One\n", + "starts with an initial guess $\\mathbf{x}_0$ for a minimum of $F$ and\n", + "computes new approximations according to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{x}_{k+1} = \\mathbf{x}_k - \\gamma_k \\nabla F(\\mathbf{x}_k), \\ \\ k \\geq 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The parameter $\\gamma_k$ is often referred to as the step length or\n", + "the learning rate within the context of Machine Learning.\n", + "\n", + "\n", + "## The ideal\n", + "\n", + "Ideally the sequence $\\{\\mathbf{x}_k \\}_{k=0}$ converges to a global\n", + "minimum of the function $F$. In general we do not know if we are in a\n", + "global or local minimum. In the special case when $F$ is a convex\n", + "function, all local minima are also global minima, so in this case\n", + "gradient descent can converge to the global solution. The advantage of\n", + "this scheme is that it is conceptually simple and straightforward to\n", + "implement. However the method in this form has some severe\n", + "limitations:\n", + "\n", + "In machine learing we are often faced with non-convex high dimensional\n", + "cost functions with many local minima. Since GD is deterministic we\n", + "will get stuck in a local minimum, if the method converges, unless we\n", + "have a very good intial guess. This also implies that the scheme is\n", + "sensitive to the chosen initial condition.\n", + "\n", + "Note that the gradient is a function of $\\mathbf{x} =\n", + "(x_1,\\cdots,x_n)$ which makes it expensive to compute numerically.\n", + "\n", + "\n", + "\n", + "## The sensitiveness of the gradient descent\n", + "\n", + "The gradient descent method \n", + "is sensitive to the choice of learning rate $\\gamma_k$. This is due\n", + "to the fact that we are only guaranteed that $F(\\mathbf{x}_{k+1}) \\leq\n", + "F(\\mathbf{x}_k)$ for sufficiently small $\\gamma_k$. The problem is to\n", + "determine an optimal learning rate. If the learning rate is chosen too\n", + "small the method will take a long time to converge and if it is too\n", + "large we can experience erratic behavior.\n", + "\n", + "Many of these shortcomings can be alleviated by introducing\n", + "randomness. One such method is that of Stochastic Gradient Descent\n", + "(SGD), see below.\n", + "\n", + "\n", + "\n", + "## Convex functions\n", + "\n", + "Ideally we want our cost/loss function to be convex(concave).\n", + "\n", + "First we give the definition of a convex set: A set $C$ in\n", + "$\\mathbb{R}^n$ is said to be convex if, for all $x$ and $y$ in $C$ and\n", + "all $t \\in (0,1)$ , the point $(1 − t)x + ty$ also belongs to\n", + "C. Geometrically this means that every point on the line segment\n", + "connecting $x$ and $y$ is in $C$ as discussed below.\n", + "\n", + "The convex subsets of $\\mathbb{R}$ are the intervals of\n", + "$\\mathbb{R}$. Examples of convex sets of $\\mathbb{R}^2$ are the\n", + "regular polygons (triangles, rectangles, pentagons, etc...).\n", + "\n", + "\n", + "## Convex function\n", + "\n", + "**Convex function**: Let $X \\subset \\mathbb{R}^n$ be a convex set. Assume that the function $f: X \\rightarrow \\mathbb{R}$ is continuous, then $f$ is said to be convex if $$f(tx_1 + (1-t)x_2) \\leq tf(x_1) + (1-t)f(x_2) $$ for all $x_1, x_2 \\in X$ and for all $t \\in [0,1]$. If $\\leq$ is replaced with a strict inequaltiy in the definition, we demand $x_1 \\neq x_2$ and $t\\in(0,1)$ then $f$ is said to be strictly convex. For a single variable function, convexity means that if you draw a straight line connecting $f(x_1)$ and $f(x_2)$, the value of the function on the interval $[x_1,x_2]$ is always below the line as illustrated below.\n", + "\n", + "\n", + "## Conditions on convex functions\n", + "\n", + "In the following we state first and second-order conditions which\n", + "ensures convexity of a function $f$. We write $D_f$ to denote the\n", + "domain of $f$, i.e the subset of $R^n$ where $f$ is defined. For more\n", + "details and proofs we refer to: [S. Boyd and L. Vandenberghe. Convex Optimization. Cambridge University Press](http://stanford.edu/boyd/cvxbook/, 2004).\n", + "\n", + "**First order condition.**\n", + "\n", + "Suppose $f$ is differentiable (i.e $\\nabla f(x)$ is well defined for\n", + "all $x$ in the domain of $f$). Then $f$ is convex if and only if $D_f$\n", + "is a convex set and $$f(y) \\geq f(x) + \\nabla f(x)^T (y-x) $$ holds\n", + "for all $x,y \\in D_f$. This condition means that for a convex function\n", + "the first order Taylor expansion (right hand side above) at any point\n", + "a global under estimator of the function. To convince yourself you can\n", + "make a drawing of $f(x) = x^2+1$ and draw the tangent line to $f(x)$ and\n", + "note that it is always below the graph.\n", + "\n", + "\n", + "\n", + "**Second order condition.**\n", + "\n", + "Assume that $f$ is twice\n", + "differentiable, i.e the Hessian matrix exists at each point in\n", + "$D_f$. Then $f$ is convex if and only if $D_f$ is a convex set and its\n", + "Hessian is positive semi-definite for all $x\\in D_f$. For a\n", + "single-variable function this reduces to $f''(x) \\geq 0$. Geometrically this means that $f$ has nonnegative curvature\n", + "everywhere.\n", + "\n", + "\n", + "\n", + "This condition is particularly useful since it gives us an procedure for determining if the function under consideration is convex, apart from using the definition.\n", + "\n", + "\n", + "## More on convex functions\n", + "\n", + "The next result is of great importance to us and the reason why we are\n", + "going on about convex functions. In machine learning we frequently\n", + "have to minimize a loss/cost function in order to find the best\n", + "parameters for the model we are considering. \n", + "\n", + "Ideally we want the\n", + "global minimum (for high-dimensional models it is hard to know\n", + "if we have local or global minimum). However, if the cost/loss function\n", + "is convex the following result provides invaluable information:\n", + "\n", + "**Any minimum is global for convex functions.**\n", + "\n", + "Consider the problem of finding $x \\in \\mathbb{R}^n$ such that $f(x)$\n", + "is minimal, where $f$ is convex and differentiable. Then, any point\n", + "$x^*$ that satisfies $\\nabla f(x^*) = 0$ is a global minimum.\n", + "\n", + "\n", + "\n", + "This result means that if we know that the cost/loss function is convex and we are able to find a minimum, we are guaranteed that it is a global minimum.\n", + "\n", + "\n", + "## Some simple problems\n", + "\n", + "1. Show that $f(x)=x^2$ is convex for $x \\in \\mathbb{R}$ using the definition of convexity. Hint: If you re-write the definition, $f$ is convex if the following holds for all $x,y \\in D_f$ and any $\\lambda \\in [0,1]$ $\\lambda f(x)+(1-\\lambda)f(y)-f(\\lambda x + (1-\\lambda) y ) \\geq 0$.\n", + "\n", + "2. Using the second order condition show that the following functions are convex on the specified domain.\n", + "\n", + " * $f(x) = e^x$ is convex for $x \\in \\mathbb{R}$.\n", + "\n", + " * $g(x) = -\\ln(x)$ is convex for $x \\in (0,\\infty)$.\n", + "\n", + "\n", + "3. Let $f(x) = x^2$ and $g(x) = e^x$. Show that $f(g(x))$ and $g(f(x))$ is convex for $x \\in \\mathbb{R}$. Also show that if $f(x)$ is any convex function than $h(x) = e^{f(x)}$ is convex.\n", + "\n", + "4. A norm is any function that satisfy the following properties\n", + "\n", + " * $f(\\alpha x) = |\\alpha| f(x)$ for all $\\alpha \\in \\mathbb{R}$.\n", + "\n", + " * $f(x+y) \\leq f(x) + f(y)$\n", + "\n", + " * $f(x) \\leq 0$ for all $x \\in \\mathbb{R}^n$ with equality if and only if $x = 0$\n", + "\n", + "\n", + "Using the definition of convexity, try to show that a function satisfying the properties above is convex (the third condition is not needed to show this).\n", + "\n", + "\n", + "\n", + "## Friday September 25\n", + "\n", + "[Video of Lecture](https://www.uio.no/studier/emner/matnat/fys/FYS-STK4155/h20/forelesningsvideoer/LectureSeptember25.mp4?vrtx=view-as-webpage) and [link to handwritten notes](https://github.com/CompPhysics/MachineLearning/blob/master/doc/HandWrittenNotes/NotesSeptember25.pdf).\n", + "\n", + "\n", + "\n", + "## Standard steepest descent\n", + "\n", + "\n", + "Before we proceed, we would like to discuss the approach called the\n", + "**standard Steepest descent** (different from the above steepest descent discussion), which again leads to us having to be able\n", + "to compute a matrix. It belongs to the class of Conjugate Gradient methods (CG).\n", + "\n", + "[The success of the CG method](https://www.cs.cmu.edu/~quake-papers/painless-conjugate-gradient.pdf)\n", + "for finding solutions of non-linear problems is based on the theory\n", + "of conjugate gradients for linear systems of equations. It belongs to\n", + "the class of iterative methods for solving problems from linear\n", + "algebra of the type" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{x} = \\boldsymbol{b}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the iterative process we end up with a problem like" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}= \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $\\boldsymbol{r}$ is the so-called residual or error in the iterative process.\n", + "\n", + "When we have found the exact solution, $\\boldsymbol{r}=0$.\n", + "\n", + "\n", + "## Gradient method\n", + "\n", + "The residual is zero when we reach the minimum of the quadratic equation" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "P(\\boldsymbol{x})=\\frac{1}{2}\\boldsymbol{x}^T\\boldsymbol{A}\\boldsymbol{x} - \\boldsymbol{x}^T\\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with the constraint that the matrix $\\boldsymbol{A}$ is positive definite and\n", + "symmetric. This defines also the Hessian and we want it to be positive definite. \n", + "\n", + "\n", + "\n", + "## Steepest descent method\n", + "\n", + "We denote the initial guess for $\\boldsymbol{x}$ as $\\boldsymbol{x}_0$. \n", + "We can assume without loss of generality that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_0=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or consider the system" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{z} = \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "instead.\n", + "\n", + "\n", + "\n", + "## Steepest descent method\n", + "One can show that the solution $\\boldsymbol{x}$ is also the unique minimizer of the quadratic form" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(\\boldsymbol{x}) = \\frac{1}{2}\\boldsymbol{x}^T\\boldsymbol{A}\\boldsymbol{x} - \\boldsymbol{x}^T \\boldsymbol{x} , \\quad \\boldsymbol{x}\\in\\mathbf{R}^n.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This suggests taking the first basis vector $\\boldsymbol{r}_1$ (see below for definition) \n", + "to be the gradient of $f$ at $\\boldsymbol{x}=\\boldsymbol{x}_0$, \n", + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{x}_0-\\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and \n", + "$\\boldsymbol{x}_0=0$ it is equal $-\\boldsymbol{b}$.\n", + "\n", + "\n", + "\n", + "\n", + "## Final expressions\n", + "We can compute the residual iteratively as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_{k+1}=\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_{k+1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{b}-\\boldsymbol{A}(\\boldsymbol{x}_k+\\alpha_k\\boldsymbol{r}_k),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_k)-\\alpha_k\\boldsymbol{A}\\boldsymbol{r}_k,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which gives" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\alpha_k = \\frac{\\boldsymbol{r}_k^T\\boldsymbol{r}_k}{\\boldsymbol{r}_k^T\\boldsymbol{A}\\boldsymbol{r}_k}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "leading to the iterative scheme" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_{k+1}=\\boldsymbol{x}_k-\\alpha_k\\boldsymbol{r}_{k},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Steepest descent example" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "import numpy.linalg as la\n", + "\n", + "import scipy.optimize as sopt\n", + "\n", + "import matplotlib.pyplot as pt\n", + "from mpl_toolkits.mplot3d import axes3d\n", + "\n", + "def f(x):\n", + " return 0.5*x[0]**2 + 2.5*x[1]**2\n", + "\n", + "def df(x):\n", + " return np.array([x[0], 5*x[1]])\n", + "\n", + "fig = pt.figure()\n", + "ax = fig.gca(projection=\"3d\")\n", + "\n", + "xmesh, ymesh = np.mgrid[-2:2:50j,-2:2:50j]\n", + "fmesh = f(np.array([xmesh, ymesh]))\n", + "ax.plot_surface(xmesh, ymesh, fmesh)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "And then as countor plot" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "pt.axis(\"equal\")\n", + "pt.contour(xmesh, ymesh, fmesh)\n", + "guesses = [np.array([2, 2./5])]" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Find guesses" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "x = guesses[-1]\n", + "s = -df(x)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Run it!" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "def f1d(alpha):\n", + " return f(x + alpha*s)\n", + "\n", + "alpha_opt = sopt.golden(f1d)\n", + "next_guess = x + alpha_opt * s\n", + "guesses.append(next_guess)\n", + "print(next_guess)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "What happened?" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "pt.axis(\"equal\")\n", + "pt.contour(xmesh, ymesh, fmesh, 50)\n", + "it_array = np.array(guesses)\n", + "pt.plot(it_array.T[0], it_array.T[1], \"x-\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method\n", + "In the CG method we define so-called conjugate directions and two vectors \n", + "$\\boldsymbol{s}$ and $\\boldsymbol{t}$\n", + "are said to be\n", + "conjugate if" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{s}^T\\boldsymbol{A}\\boldsymbol{t}= 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The philosophy of the CG method is to perform searches in various conjugate directions\n", + "of our vectors $\\boldsymbol{x}_i$ obeying the above criterion, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_i^T\\boldsymbol{A}\\boldsymbol{x}_j= 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Two vectors are conjugate if they are orthogonal with respect to \n", + "this inner product. Being conjugate is a symmetric relation: if $\\boldsymbol{s}$ is conjugate to $\\boldsymbol{t}$, then $\\boldsymbol{t}$ is conjugate to $\\boldsymbol{s}$.\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "An example is given by the eigenvectors of the matrix" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{v}_i^T\\boldsymbol{A}\\boldsymbol{v}_j= \\lambda\\boldsymbol{v}_i^T\\boldsymbol{v}_j,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which is zero unless $i=j$.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "Assume now that we have a symmetric positive-definite matrix $\\boldsymbol{A}$ of size\n", + "$n\\times n$. At each iteration $i+1$ we obtain the conjugate direction of a vector" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_{i+1}=\\boldsymbol{x}_{i}+\\alpha_i\\boldsymbol{p}_{i}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We assume that $\\boldsymbol{p}_{i}$ is a sequence of $n$ mutually conjugate directions. \n", + "Then the $\\boldsymbol{p}_{i}$ form a basis of $R^n$ and we can expand the solution \n", + "$ \\boldsymbol{A}\\boldsymbol{x} = \\boldsymbol{b}$ in this basis, namely" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x} = \\sum^{n}_{i=1} \\alpha_i \\boldsymbol{p}_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method\n", + "The coefficients are given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\mathbf{A}\\mathbf{x} = \\sum^{n}_{i=1} \\alpha_i \\mathbf{A} \\mathbf{p}_i = \\mathbf{b}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Multiplying with $\\boldsymbol{p}_k^T$ from the left gives" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{p}_k^T \\boldsymbol{A}\\boldsymbol{x} = \\sum^{n}_{i=1} \\alpha_i\\boldsymbol{p}_k^T \\boldsymbol{A}\\boldsymbol{p}_i= \\boldsymbol{p}_k^T \\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and we can define the coefficients $\\alpha_k$ as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\alpha_k = \\frac{\\boldsymbol{p}_k^T \\boldsymbol{b}}{\\boldsymbol{p}_k^T \\boldsymbol{A} \\boldsymbol{p}_k}\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method and iterations\n", + "\n", + "If we choose the conjugate vectors $\\boldsymbol{p}_k$ carefully, \n", + "then we may not need all of them to obtain a good approximation to the solution \n", + "$\\boldsymbol{x}$. \n", + "We want to regard the conjugate gradient method as an iterative method. \n", + "This will us to solve systems where $n$ is so large that the direct \n", + "method would take too much time.\n", + "\n", + "We denote the initial guess for $\\boldsymbol{x}$ as $\\boldsymbol{x}_0$. \n", + "We can assume without loss of generality that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{x}_0=0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or consider the system" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{z} = \\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_0,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "instead.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "One can show that the solution $\\boldsymbol{x}$ is also the unique minimizer of the quadratic form" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "f(\\boldsymbol{x}) = \\frac{1}{2}\\boldsymbol{x}^T\\boldsymbol{A}\\boldsymbol{x} - \\boldsymbol{x}^T \\boldsymbol{x} , \\quad \\boldsymbol{x}\\in\\mathbf{R}^n.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This suggests taking the first basis vector $\\boldsymbol{p}_1$ \n", + "to be the gradient of $f$ at $\\boldsymbol{x}=\\boldsymbol{x}_0$, \n", + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{A}\\boldsymbol{x}_0-\\boldsymbol{b},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and \n", + "$\\boldsymbol{x}_0=0$ it is equal $-\\boldsymbol{b}$.\n", + "The other vectors in the basis will be conjugate to the gradient, \n", + "hence the name conjugate gradient method.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Conjugate gradient method\n", + "Let $\\boldsymbol{r}_k$ be the residual at the $k$-th step:" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_k=\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_k.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Note that $\\boldsymbol{r}_k$ is the negative gradient of $f$ at \n", + "$\\boldsymbol{x}=\\boldsymbol{x}_k$, \n", + "so the gradient descent method would be to move in the direction $\\boldsymbol{r}_k$. \n", + "Here, we insist that the directions $\\boldsymbol{p}_k$ are conjugate to each other, \n", + "so we take the direction closest to the gradient $\\boldsymbol{r}_k$ \n", + "under the conjugacy constraint. \n", + "This gives the following expression" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{p}_{k+1}=\\boldsymbol{r}_k-\\frac{\\boldsymbol{p}_k^T \\boldsymbol{A}\\boldsymbol{r}_k}{\\boldsymbol{p}_k^T\\boldsymbol{A}\\boldsymbol{p}_k} \\boldsymbol{p}_k.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conjugate gradient method\n", + "We can also compute the residual iteratively as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_{k+1}=\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_{k+1},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which equals" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{b}-\\boldsymbol{A}(\\boldsymbol{x}_k+\\alpha_k\\boldsymbol{p}_k),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "or" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "(\\boldsymbol{b}-\\boldsymbol{A}\\boldsymbol{x}_k)-\\alpha_k\\boldsymbol{A}\\boldsymbol{p}_k,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "which gives" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{r}_{k+1}=\\boldsymbol{r}_k-\\boldsymbol{A}\\boldsymbol{p}_{k},\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Revisiting our first homework\n", + "\n", + "We will use linear regression as a case study for the gradient descent\n", + "methods. Linear regression is a great test case for the gradient\n", + "descent methods discussed in the lectures since it has several\n", + "desirable properties such as:\n", + "\n", + "1. An analytical solution (recall homework set 1).\n", + "\n", + "2. The gradient can be computed analytically.\n", + "\n", + "3. The cost function is convex which guarantees that gradient descent converges for small enough learning rates\n", + "\n", + "We revisit an example similar to what we had in the first homework set. We had a function of the type" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "x = 2*np.random.rand(m,1)\n", + "y = 4+3*x+np.random.randn(m,1)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "with $x_i \\in [0,1] $ is chosen randomly using a uniform distribution. Additionally we have a stochastic noise chosen according to a normal distribution $\\cal {N}(0,1)$. \n", + "The linear regression model is given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "h_\\beta(x) = \\boldsymbol{y} = \\beta_0 + \\beta_1 x,\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "such that" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{y}_i = \\beta_0 + \\beta_1 x_i.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Gradient descent example\n", + "\n", + "Let $\\mathbf{y} = (y_1,\\cdots,y_n)^T$, $\\mathbf{\\boldsymbol{y}} = (\\boldsymbol{y}_1,\\cdots,\\boldsymbol{y}_n)^T$ and $\\beta = (\\beta_0, \\beta_1)^T$\n", + "\n", + "It is convenient to write $\\mathbf{\\boldsymbol{y}} = X\\beta$ where $X \\in \\mathbb{R}^{100 \\times 2} $ is the design matrix given by (we keep the intercept here)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "X \\equiv \\begin{bmatrix}\n", + "1 & x_1 \\\\\n", + "\\vdots & \\vdots \\\\\n", + "1 & x_{100} & \\\\\n", + "\\end{bmatrix}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "The cost/loss/risk function is given by (" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\beta) = \\frac{1}{n}||X\\beta-\\mathbf{y}||_{2}^{2} = \\frac{1}{n}\\sum_{i=1}^{100}\\left[ (\\beta_0 + \\beta_1 x_i)^2 - 2 y_i (\\beta_0 + \\beta_1 x_i) + y_i^2\\right]\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "and we want to find $\\beta$ such that $C(\\beta)$ is minimized.\n", + "\n", + "\n", + "## The derivative of the cost/loss function\n", + "\n", + "Computing $\\partial C(\\beta) / \\partial \\beta_0$ and $\\partial C(\\beta) / \\partial \\beta_1$ we can show that the gradient can be written as" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_{\\beta} C(\\beta) = \\frac{2}{n}\\begin{bmatrix} \\sum_{i=1}^{100} \\left(\\beta_0+\\beta_1x_i-y_i\\right) \\\\\n", + "\\sum_{i=1}^{100}\\left( x_i (\\beta_0+\\beta_1x_i)-y_ix_i\\right) \\\\\n", + "\\end{bmatrix} = \\frac{2}{n}X^T(X\\beta - \\mathbf{y}),\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $X$ is the design matrix defined above.\n", + "\n", + "\n", + "## The Hessian matrix\n", + "The Hessian matrix of $C(\\beta)$ is given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\boldsymbol{H} \\equiv \\begin{bmatrix}\n", + "\\frac{\\partial^2 C(\\beta)}{\\partial \\beta_0^2} & \\frac{\\partial^2 C(\\beta)}{\\partial \\beta_0 \\partial \\beta_1} \\\\\n", + "\\frac{\\partial^2 C(\\beta)}{\\partial \\beta_0 \\partial \\beta_1} & \\frac{\\partial^2 C(\\beta)}{\\partial \\beta_1^2} & \\\\\n", + "\\end{bmatrix} = \\frac{2}{n}X^T X.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "This result implies that $C(\\beta)$ is a convex function since the matrix $X^T X$ always is positive semi-definite.\n", + "\n", + "\n", + "\n", + "\n", + "\n", + "## Simple program\n", + "\n", + "We can now write a program that minimizes $C(\\beta)$ using the gradient descent method with a constant learning rate $\\gamma$ according to" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_{k+1} = \\beta_k - \\gamma \\nabla_\\beta C(\\beta_k), \\ k=0,1,\\cdots\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can use the expression we computed for the gradient and let use a\n", + "$\\beta_0$ be chosen randomly and let $\\gamma = 0.001$. Stop iterating\n", + "when $||\\nabla_\\beta C(\\beta_k) || \\leq \\epsilon = 10^{-8}$. **Note that the code below does not include the latter stop criterion**.\n", + "\n", + "And finally we can compare our solution for $\\beta$ with the analytic result given by \n", + "$\\beta= (X^TX)^{-1} X^T \\mathbf{y}$.\n", + "\n", + "\n", + "## Gradient Descent Example\n", + "\n", + "Here our simple example" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "\n", + "# Importing various packages\n", + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from mpl_toolkits.mplot3d import Axes3D\n", + "from matplotlib import cm\n", + "from matplotlib.ticker import LinearLocator, FormatStrFormatter\n", + "import sys\n", + "\n", + "# the number of datapoints\n", + "n = 100\n", + "x = 2*np.random.rand(n,1)\n", + "y = 4+3*x+np.random.randn(n,1)\n", + "\n", + "X = np.c_[np.ones((n,1)), x]\n", + "# Hessian matrix\n", + "H = (2.0/n)* X.T @ X\n", + "# Get the eigenvalues\n", + "EigValues, EigVectors = np.linalg.eig(H)\n", + "print(EigValues)\n", + "\n", + "beta_linreg = np.linalg.inv(X.T @ X) @ X.T @ y\n", + "print(beta_linreg)\n", + "beta = np.random.randn(2,1)\n", + "\n", + "eta = 1.0/np.max(EigValues)\n", + "Niterations = 1000\n", + "\n", + "for iter in range(Niterations):\n", + " gradient = (2.0/n)*X.T @ (X @ beta-y)\n", + " beta -= eta*gradient\n", + "\n", + "print(beta)\n", + "xnew = np.array([[0],[2]])\n", + "xbnew = np.c_[np.ones((2,1)), xnew]\n", + "ypredict = xbnew.dot(beta)\n", + "ypredict2 = xbnew.dot(beta_linreg)\n", + "plt.plot(xnew, ypredict, \"r-\")\n", + "plt.plot(xnew, ypredict2, \"b-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,2.0,0, 15.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Gradient descent example')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## And a corresponding example using **scikit-learn**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import SGDRegressor\n", + "\n", + "n = 100\n", + "x = 2*np.random.rand(n,1)\n", + "y = 4+3*x+np.random.randn(n,1)\n", + "\n", + "X = np.c_[np.ones((n,1)), x]\n", + "beta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)\n", + "print(beta_linreg)\n", + "sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1)\n", + "sgdreg.fit(x,y.ravel())\n", + "print(sgdreg.intercept_, sgdreg.coef_)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Gradient descent and Ridge\n", + "\n", + "We have also discussed Ridge regression where the loss function contains a regularized term given by the $L_2$ norm of $\\beta$," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C_{\\text{ridge}}(\\beta) = \\frac{1}{n}||X\\beta -\\mathbf{y}||^2 + \\lambda ||\\beta||^2, \\ \\lambda \\geq 0.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In order to minimize $C_{\\text{ridge}}(\\beta)$ using GD we only have adjust the gradient as follows" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_\\beta C_{\\text{ridge}}(\\beta) = \\frac{2}{n}\\begin{bmatrix} \\sum_{i=1}^{100} \\left(\\beta_0+\\beta_1x_i-y_i\\right) \\\\\n", + "\\sum_{i=1}^{100}\\left( x_i (\\beta_0+\\beta_1x_i)-y_ix_i\\right) \\\\\n", + "\\end{bmatrix} + 2\\lambda\\begin{bmatrix} \\beta_0 \\\\ \\beta_1\\end{bmatrix} = 2 (X^T(X\\beta - \\mathbf{y})+\\lambda \\beta).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "We can easily extend our program to minimize $C_{\\text{ridge}}(\\beta)$ using gradient descent and compare with the analytical solution given by" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_{\\text{ridge}} = \\left(X^T X + \\lambda I_{2 \\times 2} \\right)^{-1} X^T \\mathbf{y}.\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Program example for gradient descent with Ridge Regression" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from mpl_toolkits.mplot3d import Axes3D\n", + "from matplotlib import cm\n", + "from matplotlib.ticker import LinearLocator, FormatStrFormatter\n", + "import sys\n", + "\n", + "# the number of datapoints\n", + "n = 100\n", + "x = 2*np.random.rand(n,1)\n", + "y = 4+3*x+np.random.randn(n,1)\n", + "\n", + "X = np.c_[np.ones((n,1)), x]\n", + "XT_X = X.T @ X\n", + "\n", + "#Ridge parameter lambda\n", + "lmbda = 0.001\n", + "Id = lmbda* np.eye(XT_X.shape[0])\n", + "\n", + "beta_linreg = np.linalg.inv(XT_X+Id) @ X.T @ y\n", + "print(beta_linreg)\n", + "# Start plain gradient descent\n", + "beta = np.random.randn(2,1)\n", + "\n", + "eta = 0.1\n", + "Niterations = 100\n", + "\n", + "for iter in range(Niterations):\n", + " gradients = 2.0/n*X.T @ (X @ (beta)-y)+2*lmbda*beta\n", + " beta -= eta*gradients\n", + "\n", + "print(beta)\n", + "ypredict = X @ beta\n", + "ypredict2 = X @ beta_linreg\n", + "plt.plot(x, ypredict, \"r-\")\n", + "plt.plot(x, ypredict2, \"b-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,2.0,0, 15.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Gradient descent example for Ridge')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Using gradient descent methods, limitations\n", + "\n", + "* **Gradient descent (GD) finds local minima of our function**. Since the GD algorithm is deterministic, if it converges, it will converge to a local minimum of our cost/loss/risk function. Because in ML we are often dealing with extremely rugged landscapes with many local minima, this can lead to poor performance.\n", + "\n", + "* **GD is sensitive to initial conditions**. One consequence of the local nature of GD is that initial conditions matter. Depending on where one starts, one will end up at a different local minima. Therefore, it is very important to think about how one initializes the training process. This is true for GD as well as more complicated variants of GD.\n", + "\n", + "* **Gradients are computationally expensive to calculate for large datasets**. In many cases in statistics and ML, the cost/loss/risk function is a sum of terms, with one term for each data point. For example, in linear regression, $E \\propto \\sum_{i=1}^n (y_i - \\mathbf{w}^T\\cdot\\mathbf{x}_i)^2$; for logistic regression, the square error is replaced by the cross entropy. To calculate the gradient we have to sum over *all* $n$ data points. Doing this at every GD step becomes extremely computationally expensive. An ingenious solution to this, is to calculate the gradients using small subsets of the data called \"mini batches\". This has the added benefit of introducing stochasticity into our algorithm.\n", + "\n", + "* **GD is very sensitive to choices of learning rates**. GD is extremely sensitive to the choice of learning rates. If the learning rate is very small, the training process take an extremely long time. For larger learning rates, GD can diverge and give poor results. Furthermore, depending on what the local landscape looks like, we have to modify the learning rates to ensure convergence. Ideally, we would *adaptively* choose the learning rates to match the landscape.\n", + "\n", + "* **GD treats all directions in parameter space uniformly.** Another major drawback of GD is that unlike Newton's method, the learning rate for GD is the same in all directions in parameter space. For this reason, the maximum learning rate is set by the behavior of the steepest direction and this can significantly slow down training. Ideally, we would like to take large steps in flat directions and small steps in steep directions. Since we are exploring rugged landscapes where curvatures change, this requires us to keep track of not only the gradient but second derivatives. The ideal scenario would be to calculate the Hessian but this proves to be too computationally expensive. \n", + "\n", + "* GD can take exponential time to escape saddle points, even with random initialization. As we mentioned, GD is extremely sensitive to initial condition since it determines the particular local minimum GD would eventually reach. However, even with a good initialization scheme, through the introduction of randomness, GD can still take exponential time to escape saddle points.\n", + "\n", + "## Stochastic Gradient Descent\n", + "\n", + "Stochastic gradient descent (SGD) and variants thereof address some of\n", + "the shortcomings of the Gradient descent method discussed above.\n", + "\n", + "The underlying idea of SGD comes from the observation that the cost\n", + "function, which we want to minimize, can almost always be written as a\n", + "sum over $n$ data points $\\{\\mathbf{x}_i\\}_{i=1}^n$," + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "C(\\mathbf{\\beta}) = \\sum_{i=1}^n c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Computation of gradients\n", + "\n", + "This in turn means that the gradient can be\n", + "computed as a sum over $i$-gradients" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_\\beta C(\\mathbf{\\beta}) = \\sum_i^n \\nabla_\\beta c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Stochasticity/randomness is introduced by only taking the\n", + "gradient on a subset of the data called minibatches. If there are $n$\n", + "data points and the size of each minibatch is $M$, there will be $n/M$\n", + "minibatches. We denote these minibatches by $B_k$ where\n", + "$k=1,\\cdots,n/M$.\n", + "\n", + "\n", + "## SGD example\n", + "As an example, suppose we have $10$ data points $(\\mathbf{x}_1,\\cdots, \\mathbf{x}_{10})$ \n", + "and we choose to have $M=5$ minibathces,\n", + "then each minibatch contains two data points. In particular we have\n", + "$B_1 = (\\mathbf{x}_1,\\mathbf{x}_2), \\cdots, B_5 =\n", + "(\\mathbf{x}_9,\\mathbf{x}_{10})$. Note that if you choose $M=1$ you\n", + "have only a single batch with all data points and on the other extreme,\n", + "you may choose $M=n$ resulting in a minibatch for each datapoint, i.e\n", + "$B_k = \\mathbf{x}_k$.\n", + "\n", + "The idea is now to approximate the gradient by replacing the sum over\n", + "all data points with a sum over the data points in one the minibatches\n", + "picked at random in each gradient descent step" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\nabla_{\\beta}\n", + "C(\\mathbf{\\beta}) = \\sum_{i=1}^n \\nabla_\\beta c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta}) \\rightarrow \\sum_{i \\in B_k}^n \\nabla_\\beta\n", + "c_i(\\mathbf{x}_i, \\mathbf{\\beta}).\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## The gradient step\n", + "\n", + "Thus a gradient descent step now looks like" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "$$\n", + "\\beta_{j+1} = \\beta_j - \\gamma_j \\sum_{i \\in B_k}^n \\nabla_\\beta c_i(\\mathbf{x}_i,\n", + "\\mathbf{\\beta})\n", + "$$" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "where $k$ is picked at random with equal\n", + "probability from $[1,n/M]$. An iteration over the number of\n", + "minibathces (n/M) is commonly referred to as an epoch. Thus it is\n", + "typical to choose a number of epochs and for each epoch iterate over\n", + "the number of minibatches, as exemplified in the code below.\n", + "\n", + "\n", + "## Simple example code" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np \n", + "\n", + "n = 100 #100 datapoints \n", + "M = 5 #size of each minibatch\n", + "m = int(n/M) #number of minibatches\n", + "n_epochs = 10 #number of epochs\n", + "\n", + "j = 0\n", + "for epoch in range(1,n_epochs+1):\n", + " for i in range(m):\n", + " k = np.random.randint(m) #Pick the k-th minibatch at random\n", + " #Compute the gradient using the data in minibatch Bk\n", + " #Compute new suggestion for \n", + " j += 1" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Taking the gradient only on a subset of the data has two important\n", + "benefits. First, it introduces randomness which decreases the chance\n", + "that our opmization scheme gets stuck in a local minima. Second, if\n", + "the size of the minibatches are small relative to the number of\n", + "datapoints ($M < n$), the computation of the gradient is much\n", + "cheaper since we sum over the datapoints in the $k-th$ minibatch and not\n", + "all $n$ datapoints.\n", + "\n", + "\n", + "## When do we stop?\n", + "\n", + "A natural question is when do we stop the search for a new minimum?\n", + "One possibility is to compute the full gradient after a given number\n", + "of epochs and check if the norm of the gradient is smaller than some\n", + "threshold and stop if true. However, the condition that the gradient\n", + "is zero is valid also for local minima, so this would only tell us\n", + "that we are close to a local/global minimum. However, we could also\n", + "evaluate the cost function at this point, store the result and\n", + "continue the search. If the test kicks in at a later stage we can\n", + "compare the values of the cost function and keep the $\\beta$ that\n", + "gave the lowest value.\n", + "\n", + "\n", + "## Slightly different approach\n", + "\n", + "Another approach is to let the step length $\\gamma_j$ depend on the\n", + "number of epochs in such a way that it becomes very small after a\n", + "reasonable time such that we do not move at all.\n", + "\n", + "As an example, let $e = 0,1,2,3,\\cdots$ denote the current epoch and let $t_0, t_1 > 0$ be two fixed numbers. Furthermore, let $t = e \\cdot m + i$ where $m$ is the number of minibatches and $i=0,\\cdots,m-1$. Then the function $$\\gamma_j(t; t_0, t_1) = \\frac{t_0}{t+t_1} $$ goes to zero as the number of epochs gets large. I.e. we start with a step length $\\gamma_j (0; t_0, t_1) = t_0/t_1$ which decays in *time* $t$.\n", + "\n", + "In this way we can fix the number of epochs, compute $\\beta$ and\n", + "evaluate the cost function at the end. Repeating the computation will\n", + "give a different result since the scheme is random by design. Then we\n", + "pick the final $\\beta$ that gives the lowest value of the cost\n", + "function." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "import numpy as np \n", + "\n", + "def step_length(t,t0,t1):\n", + " return t0/(t+t1)\n", + "\n", + "n = 100 #100 datapoints \n", + "M = 5 #size of each minibatch\n", + "m = int(n/M) #number of minibatches\n", + "n_epochs = 500 #number of epochs\n", + "t0 = 1.0\n", + "t1 = 10\n", + "\n", + "gamma_j = t0/t1\n", + "j = 0\n", + "for epoch in range(1,n_epochs+1):\n", + " for i in range(m):\n", + " k = np.random.randint(m) #Pick the k-th minibatch at random\n", + " #Compute the gradient using the data in minibatch Bk\n", + " #Compute new suggestion for beta\n", + " t = epoch*m+i\n", + " gamma_j = step_length(t,t0,t1)\n", + " j += 1\n", + "\n", + "print(\"gamma_j after %d epochs: %g\" % (n_epochs,gamma_j))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Program for stochastic gradient" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "collapsed": false, + "editable": true + }, + "outputs": [], + "source": [ + "# Importing various packages\n", + "from math import exp, sqrt\n", + "from random import random, seed\n", + "import numpy as np\n", + "import matplotlib.pyplot as plt\n", + "from sklearn.linear_model import SGDRegressor\n", + "\n", + "m = 100\n", + "x = 2*np.random.rand(m,1)\n", + "y = 4+3*x+np.random.randn(m,1)\n", + "\n", + "X = np.c_[np.ones((m,1)), x]\n", + "theta_linreg = np.linalg.inv(X.T @ X) @ (X.T @ y)\n", + "print(\"Own inversion\")\n", + "print(theta_linreg)\n", + "sgdreg = SGDRegressor(max_iter = 50, penalty=None, eta0=0.1)\n", + "sgdreg.fit(x,y.ravel())\n", + "print(\"sgdreg from scikit\")\n", + "print(sgdreg.intercept_, sgdreg.coef_)\n", + "\n", + "\n", + "theta = np.random.randn(2,1)\n", + "eta = 0.1\n", + "Niterations = 1000\n", + "\n", + "\n", + "for iter in range(Niterations):\n", + " gradients = 2.0/m*X.T @ ((X @ theta)-y)\n", + " theta -= eta*gradients\n", + "print(\"theta from own gd\")\n", + "print(theta)\n", + "\n", + "xnew = np.array([[0],[2]])\n", + "Xnew = np.c_[np.ones((2,1)), xnew]\n", + "ypredict = Xnew.dot(theta)\n", + "ypredict2 = Xnew.dot(theta_linreg)\n", + "\n", + "\n", + "n_epochs = 50\n", + "t0, t1 = 5, 50\n", + "def learning_schedule(t):\n", + " return t0/(t+t1)\n", + "\n", + "theta = np.random.randn(2,1)\n", + "\n", + "for epoch in range(n_epochs):\n", + " for i in range(m):\n", + " random_index = np.random.randint(m)\n", + " xi = X[random_index:random_index+1]\n", + " yi = y[random_index:random_index+1]\n", + " gradients = 2 * xi.T @ ((xi @ theta)-yi)\n", + " eta = learning_schedule(epoch*m+i)\n", + " theta = theta - eta*gradients\n", + "print(\"theta from own sdg\")\n", + "print(theta)\n", + "\n", + "plt.plot(xnew, ypredict, \"r-\")\n", + "plt.plot(xnew, ypredict2, \"b-\")\n", + "plt.plot(x, y ,'ro')\n", + "plt.axis([0,2.0,0, 15.0])\n", + "plt.xlabel(r'$x$')\n", + "plt.ylabel(r'$y$')\n", + "plt.title(r'Random numbers ')\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "**Challenge**: try to write a similar code for a Logistic Regression case." + ] + } + ], + "metadata": {}, + "nbformat": 4, + "nbformat_minor": 4 +}