diff --git a/doc/LectureNotes/fig/pandas.jpg b/doc/LectureNotes/fig/pandas.jpg new file mode 100644 index 000000000..4236914f8 Binary files /dev/null and b/doc/LectureNotes/fig/pandas.jpg differ diff --git a/doc/pub/How2ReadData/ipynb/How2ReadData.ipynb b/doc/pub/How2ReadData/ipynb/How2ReadData.ipynb index ef16f5fc9..8e59aafb4 100644 --- a/doc/pub/How2ReadData/ipynb/How2ReadData.ipynb +++ b/doc/pub/How2ReadData/ipynb/How2ReadData.ipynb @@ -2325,21 +2325,9 @@ }, { "cell_type": "code", - "execution_count": 10, + "execution_count": 12, "metadata": {}, - "outputs": [ - { - "ename": "FileNotFoundError", - "evalue": "[Errno 2] No such file or directory: 'DataFiles/MassEval2016.dat'", - "output_type": "error", - "traceback": [ - "\u001b[0;31m---------------------------------------------------------------------------\u001b[0m", - "\u001b[0;31mFileNotFoundError\u001b[0m Traceback (most recent call last)", - "\u001b[0;32m\u001b[0m in \u001b[0;36m\u001b[0;34m\u001b[0m\n\u001b[1;32m 31\u001b[0m \u001b[0mplt\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0msavefig\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mimage_path\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mfig_id\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;34m+\u001b[0m \u001b[0;34m\".png\"\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mformat\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m'png'\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m 32\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 33\u001b[0;31m \u001b[0minfile\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mopen\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mdata_path\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m\"MassEval2016.dat\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m'r'\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m", - "\u001b[0;31mFileNotFoundError\u001b[0m: [Errno 2] No such file or directory: 'DataFiles/MassEval2016.dat'" - ] - } - ], + "outputs": [], "source": [ "# Common imports\n", "import numpy as np\n", @@ -2385,7 +2373,7 @@ }, { "cell_type": "code", - "execution_count": 29, + "execution_count": 13, "metadata": {}, "outputs": [], "source": [ @@ -2446,7 +2434,7 @@ }, { "cell_type": "code", - "execution_count": 31, + "execution_count": 14, "metadata": {}, "outputs": [], "source": [ @@ -2488,9 +2476,81 @@ }, { "cell_type": "code", - "execution_count": 32, + "execution_count": 15, "metadata": {}, - "outputs": [], + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + " N Z A Element Ebinding\n", + "A \n", + "1 0 0 1 1 H 0.000000\n", + "2 1 1 1 2 H 1.112283\n", + "3 2 2 1 3 H 2.827265\n", + "4 6 2 2 4 He 7.073915\n", + "5 9 3 2 5 He 5.512132\n", + "6 14 3 3 6 Li 5.332331\n", + "7 19 4 3 7 Li 5.606439\n", + "8 24 4 4 8 Be 7.062435\n", + "9 29 5 4 9 Be 6.462668\n", + "10 34 6 4 10 Be 6.497630\n", + "11 40 6 5 11 B 6.927732\n", + "12 46 6 6 12 C 7.680144\n", + "13 52 7 6 13 C 7.469849\n", + "14 57 8 6 14 C 7.520319\n", + "15 64 8 7 15 N 7.699460\n", + "16 72 8 8 16 O 7.976206\n", + "17 78 9 8 17 O 7.750728\n", + "18 85 10 8 18 O 7.767097\n", + "19 93 10 9 19 F 7.779018\n", + "20 102 10 10 20 Ne 8.032240\n", + "21 110 11 10 21 Ne 7.971713\n", + "22 118 12 10 22 Ne 8.080465\n", + "23 128 12 11 23 Na 8.111493\n", + "24 137 12 12 24 Mg 8.260709\n", + "25 146 13 12 25 Mg 8.223502\n", + "26 154 14 12 26 Mg 8.333870\n", + "27 164 14 13 27 Al 8.331553\n", + "28 174 14 14 28 Si 8.447744\n", + "29 183 15 14 29 Si 8.448635\n", + "30 192 16 14 30 Si 8.520654\n", + "... ... ... ... ... ...\n", + "238 3089 146 92 238 U 7.570125\n", + "239 3099 146 93 239 Np 7.560567\n", + "240 3109 146 94 240 Pu 7.556042\n", + "241 3118 147 94 241 Pu 7.546439\n", + "242 3127 148 94 242 Pu 7.541327\n", + "243 3136 149 94 243 Pu 7.531008\n", + "244 3144 150 94 244 Pu 7.524815\n", + "245 3154 149 96 245 Cm 7.515767\n", + "246 3162 150 96 246 Cm 7.511471\n", + "247 3170 151 96 247 Cm 7.501931\n", + "248 3177 152 96 248 Cm 7.496728\n", + "249 3186 152 97 249 Bk 7.486040\n", + "250 3194 152 98 250 Cf 7.479956\n", + "251 3201 153 98 251 Cf 7.470500\n", + "252 3209 154 98 252 Cf 7.465347\n", + "253 3216 155 98 253 Cf 7.454829\n", + "254 3224 156 98 254 Cf 7.449225\n", + "255 3232 156 99 255 Es 7.437821\n", + "256 3241 156 100 256 Fm 7.431780\n", + "257 3248 157 100 257 Fm 7.422194\n", + "258 3256 157 101 258 Md 7.409675\n", + "259 3264 157 102 259 No 7.399974\n", + "260 3275 154 106 260 Sg 7.342562\n", + "261 3280 157 104 261 Rf 7.371384\n", + "262 3289 156 106 262 Sg 7.341185\n", + "264 3304 156 108 264 Hs 7.298375\n", + "265 3310 157 108 265 Hs 7.296247\n", + "266 3317 158 108 266 Hs 7.298273\n", + "269 3338 159 110 269 Ds 7.250154\n", + "270 3344 160 110 270 Ds 7.253775\n", + "\n", + "[267 rows x 5 columns]\n" + ] + } + ], "source": [ "A = Masses['A']\n", "Z = Masses['Z']\n", @@ -2510,7 +2570,7 @@ }, { "cell_type": "code", - "execution_count": 33, + "execution_count": 16, "metadata": {}, "outputs": [], "source": [ diff --git a/doc/pub/Regression/html/._Regression-bs000.html b/doc/pub/Regression/html/._Regression-bs000.html index 34ea38ffc..39762274c 100644 --- a/doc/pub/Regression/html/._Regression-bs000.html +++ b/doc/pub/Regression/html/._Regression-bs000.html @@ -405,7 +405,7 @@ MathJax.Hub.Config({
[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

-

Aug 19, 2019

+

Aug 21, 2019


diff --git a/doc/pub/Regression/html/._Regression-bs092.html b/doc/pub/Regression/html/._Regression-bs092.html index 1b0db8aa0..dbf917102 100644 --- a/doc/pub/Regression/html/._Regression-bs092.html +++ b/doc/pub/Regression/html/._Regression-bs092.html @@ -431,7 +431,7 @@ x_train, x_test, y_train, y_test = train_tes print('Var:', variance[degree]) print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree])) -plt.plot(polydegree, np.log10(error), label='Error') +plt.plot(polydegree, error, label='Error') plt.plot(polydegree, bias, label='bias') plt.plot(polydegree, variance, label='Variance') plt.legend() diff --git a/doc/pub/Regression/html/Regression-bs.html b/doc/pub/Regression/html/Regression-bs.html index 34ea38ffc..39762274c 100644 --- a/doc/pub/Regression/html/Regression-bs.html +++ b/doc/pub/Regression/html/Regression-bs.html @@ -405,7 +405,7 @@ MathJax.Hub.Config({

[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

-

Aug 19, 2019

+

Aug 21, 2019


diff --git a/doc/pub/Regression/html/Regression-reveal.html b/doc/pub/Regression/html/Regression-reveal.html index 71f039969..2647c3c1d 100644 --- a/doc/pub/Regression/html/Regression-reveal.html +++ b/doc/pub/Regression/html/Regression-reveal.html @@ -148,7 +148,7 @@ MathJax.Hub.Config({

[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

 
-

Aug 19, 2019

+

Aug 21, 2019


@@ -3747,7 +3747,7 @@ x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=print('Var:', variance[degree]) print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree])) -plt.plot(polydegree, np.log10(error), label='Error') +plt.plot(polydegree, error, label='Error') plt.plot(polydegree, bias, label='bias') plt.plot(polydegree, variance, label='Variance') plt.legend() diff --git a/doc/pub/Regression/html/Regression-solarized.html b/doc/pub/Regression/html/Regression-solarized.html index de85446df..ba9ade769 100644 --- a/doc/pub/Regression/html/Regression-solarized.html +++ b/doc/pub/Regression/html/Regression-solarized.html @@ -291,7 +291,7 @@ MathJax.Hub.Config({

[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

-

Aug 19, 2019

+

Aug 21, 2019












@@ -3636,7 +3636,7 @@ x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=print('Var:', variance[degree]) print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree])) -plt.plot(polydegree, np.log10(error), label='Error') +plt.plot(polydegree, error, label='Error') plt.plot(polydegree, bias, label='bias') plt.plot(polydegree, variance, label='Variance') plt.legend() diff --git a/doc/pub/Regression/html/Regression.html b/doc/pub/Regression/html/Regression.html index 7874e1d4c..17c2601d3 100644 --- a/doc/pub/Regression/html/Regression.html +++ b/doc/pub/Regression/html/Regression.html @@ -296,7 +296,7 @@ MathJax.Hub.Config({

[2] Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University

-

Aug 19, 2019

+

Aug 21, 2019












@@ -3641,7 +3641,7 @@ x_train, x_test, y_train, y_test = train_tes print('Var:', variance[degree]) print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree])) -plt.plot(polydegree, np.log10(error), label='Error') +plt.plot(polydegree, error, label='Error') plt.plot(polydegree, bias, label='bias') plt.plot(polydegree, variance, label='Variance') plt.legend() diff --git a/doc/pub/Regression/ipynb/Regression.ipynb b/doc/pub/Regression/ipynb/Regression.ipynb index 908fff4a4..0116e8535 100644 --- a/doc/pub/Regression/ipynb/Regression.ipynb +++ b/doc/pub/Regression/ipynb/Regression.ipynb @@ -10,7 +10,7 @@ " \n", "**Morten Hjorth-Jensen**, Department of Physics, University of Oslo and Department of Physics and Astronomy and National Superconducting Cyclotron Laboratory, Michigan State University\n", "\n", - "Date: **Aug 19, 2019**\n", + "Date: **Aug 21, 2019**\n", "\n", "Copyright 1999-2019, Morten Hjorth-Jensen. Released under CC Attribution-NonCommercial 4.0 license\n", "\n", @@ -4926,7 +4926,7 @@ " print('Var:', variance[degree])\n", " print('{} >= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree]))\n", "\n", - "plt.plot(polydegree, np.log10(error), label='Error')\n", + "plt.plot(polydegree, error, label='Error')\n", "plt.plot(polydegree, bias, label='bias')\n", "plt.plot(polydegree, variance, label='Variance')\n", "plt.legend()\n", diff --git a/doc/pub/Regression/ipynb/ipynb-Regression-src.tar.gz b/doc/pub/Regression/ipynb/ipynb-Regression-src.tar.gz index 40757a89d..d919eb94b 100644 Binary files a/doc/pub/Regression/ipynb/ipynb-Regression-src.tar.gz and b/doc/pub/Regression/ipynb/ipynb-Regression-src.tar.gz differ diff --git a/doc/pub/Regression/pdf/Regression-minted.pdf b/doc/pub/Regression/pdf/Regression-minted.pdf index 53db07d36..33987bff5 100644 Binary files a/doc/pub/Regression/pdf/Regression-minted.pdf and b/doc/pub/Regression/pdf/Regression-minted.pdf differ diff --git a/doc/src/Projects/2019/Exercises/clean.sh b/doc/src/Projects/2019/Exercises/clean.sh new file mode 100755 index 000000000..2e5da2c72 --- /dev/null +++ b/doc/src/Projects/2019/Exercises/clean.sh @@ -0,0 +1,3 @@ +#!/bin/sh +doconce clean +rm -rf *.pdf *.tex ipynb*.tar.gz *.html ._*.html *~ reveal.js Trash README.txt diff --git a/doc/src/Projects/2019/Exercises/hw1.do.txt b/doc/src/Projects/2019/Exercises/hw1.do.txt new file mode 100644 index 000000000..66ec64db7 --- /dev/null +++ b/doc/src/Projects/2019/Exercises/hw1.do.txt @@ -0,0 +1,157 @@ +TITLE: Homework 1 Fall Semester 2019 +AUTHOR: "Data Analysis and Machine Learning FYS-STK3155/FYS4155":"http://www.uio.no/studier/emner/matnat/fys/FYS3155/index-eng.html" {copyright, 1999-present|CC BY-NC} at Department of Physics, University of Oslo, Norway +DATE:Today + + +===== Exercise 1 ===== + +The first exercise here is of a mere technical art. We want you to have +* git as a version control software and to establish a user account on a provider like GitHub. Other providers like GitLab etc are equally fine. You can also use the University of Oslo "GitHub facilities":"https://www.uio.no/tjenester/it/maskin/filer/versjonskontroll/github.html". +* Install various Python packages + +We will make extensive use of Python as programming language and its +myriad of available libraries. You will find +IPython/Jupyter notebooks invaluable in your work. You can run _R_ +codes in the Jupyter/IPython notebooks, with the immediate benefit of +visualizing your data. You can also use compiled languages like C++, +Rust, Fortran etc if you prefer. The focus in these lectures will be +on Python. + +If you have Python installed (we recommend Python3) and you feel +pretty familiar with installing different packages, we recommend that +you install the following Python packages via _pip_ as + +o pip install numpy scipy matplotlib ipython scikit-learn sympy pandas pillow + +For _Tensorflow_, we recommend following the instructions in the text of +"Aurelien Geron, Hands‑On Machine Learning with Scikit‑Learn and TensorFlow, O'Reilly":"http://shop.oreilly.com/product/0636920052289.do" + +We will come back to _tensorflow_ later. + +For Python3, replace _pip_ with _pip3_. + +For OSX users we recommend, after having installed Xcode, to +install _brew_. Brew allows for a seamless installation of additional +software via for example + +o brew install python3 + +For Linux users, with its variety of distributions like for example the widely popular Ubuntu distribution, +you can use _pip_ as well and simply install Python as + +o sudo apt-get install python3 (or python for pyhton2.7) + +If you don't want to perform these operations separately and venture +into the hassle of exploring how to set up dependencies and paths, we +recommend two widely used distrubutions which set up all relevant +dependencies for Python, namely + +* "Anaconda":"https://docs.anaconda.com/", + +which is an open source +distribution of the Python and R programming languages for large-scale +data processing, predictive analytics, and scientific computing, that +aims to simplify package management and deployment. Package versions +are managed by the package management system _conda_. + +* "Enthought canopy":"https://www.enthought.com/product/canopy/" + +is a Python +distribution for scientific and analytic computing distribution and +analysis environment, available for free and under a commercial +license. + +We recommend using _Anaconda_. + +===== Exercise 2 ===== + +We will generate our own dataset for a function $y(x)$ where $x \in [0,1]$ and defined by random numbers computed with the uniform distribution. The function $y$ is a quadratic polynomial in $x$ with added stochastic noise according to the normal distribution $\cal {N}(0,1)$. +The following simple Python instructions define our $x$ and $y$ values (with 100 data points). +!bc pycod +x = np.random.rand(100,1) +y = 5*x*x+0.1*np.random.randn(100,1) +!ec + +o Write your own code (following the examples under the "regression slides":"https://compphysics.github.io/MachineLearning/doc/pub/Regression/html/Regression-bs.html") for computing the parametrization of the data set fitting a second-order polynomial. +o Use thereafter _scikit-learn_ (see again the examples in the regression slides) and compare with your own code. +o Using scikit-learn, compute also the mean square error, a risk metric corresponding to the expected value of the squared (quadratic) error defined as +!bt +\[ MSE(\hat{y},\hat{\tilde{y}}) = \frac{1}{n} +\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2, +\] +!et +and the $R^2$ score function. +If $\tilde{\hat{y}}_i$ is the predicted value of the $i-th$ sample and $y_i$ is the corresponding true value, then the score $R^2$ is defined as +!bt +\[ +R^2(\hat{y}, \tilde{\hat{y}}) = 1 - \frac{\sum_{i=0}^{n - 1} (y_i - \tilde{y}_i)^2}{\sum_{i=0}^{n - 1} (y_i - \bar{y})^2}, +\] +!et +where we have defined the mean value of $\hat{y}$ as +!bt +\[ +\bar{y} = \frac{1}{n} \sum_{i=0}^{n - 1} y_i. +\] +!et +You can use the functionality included in scikit-learn. If you feel for it, you can use your own program and define functions which compute the above two functions. +Discuss the meaning of these results. Try also to vary the coefficient in front of the added stochastic noise term and discuss the quality of the fits. + + + + +===== Exercise 3, mean values and variances in linear regression ===== + + +This exercise deals with various mean values ad variances in linear regression method (here it may be useful to look up chapter 3, equation (3.8) of "Trevor Hastie, Robert Tibshirani, Jerome H. Friedman, The Elements of Statistical Learning, Springer":"https://www.springer.com/gp/book/9780387848570"). + +The assumption we have made is +that there exists a function $f(\bm{x})$ and a normal distributed error $\bm{\varepsilon}\sim \mathcal{N}(0, \sigma^2)$ +which describes our data +!bt +\[ +\bm{y} = f(\bm{x})+\bm{\varepsilon} +\] +!et + +We then approximate this function with our model from the solution of the linear regression equations (ordinary least squares OLS), that is our +function $f$ is approximated by $\bm{\tilde{y}}$ where we minimized $(\bm{y}-\bm{\tilde{y}})^2$, with +!bt +\[ +\bm{\tilde{y}} = \bm{X}\bm{\beta}. +\] +!et +The matrix $\bm{X}$ is the so-called design matrix. + + +Show that the expectation value of $\bm{y}$ for a given element $i$ +!bt +\begin{align*} +\mathbb{E}(y_i) & =\mathbf{X}_{i, \ast} \, \beta, +\end{align*} +!et +and that +its variance is +!bt +\begin{align*} \mbox{Var}(y_i) & = \sigma^2. +\end{align*} +!et +Hence, $y_i \sim \mathcal{N}( \mathbf{X}_{i, \ast} \, \bm{\beta}, \sigma^2)$, that is $\bm{y}$ follows a normal distribution with +mean value $\bm{X}\bm{\beta}$ and variance $\sigma^2$. + + +With the OLS expressions for the parameters $\bm{\beta}$ show that +!bt +\[ +\mathbb{E}(\bm{\beta}) = \bm{\beta}. +\] +!et +This means that the estimator of the regression parameters is unbiased. + +Show finally that the variance of $\bm{\beta}$ is +!bt +\begin{eqnarray*} +\mbox{Var}(\bm{\beta}) & = & \sigma^2 \, (\mathbf{X}^{T} \mathbf{X})^{-1}. +\end{eqnarray*} +!et + + diff --git a/doc/src/Projects/2019/Exercises/hw2.do.txt b/doc/src/Projects/2019/Exercises/hw2.do.txt new file mode 100644 index 000000000..987cef3ec --- /dev/null +++ b/doc/src/Projects/2019/Exercises/hw2.do.txt @@ -0,0 +1,65 @@ +TITLE: Homework 2 +AUTHOR: "Data Analysis and Machine Learning FYS-STK3155/FYS4155":"http://www.uio.no/studier/emner/matnat/fys/FYS3155/index-eng.html" {copyright, 1999-present|CC BY-NC} at Department of Physics, University of Oslo, Norway +DATE:Today + + +===== Exercise 4 ===== + +This exercise is a continuation of exercise 2 from homework 1. We will +use the same function to generate our data set, still staying with a +simple function $y(x)$ which we want to fit using linear regression, +but now extending the analysis to include the Ridge and the Lasso +regression methods. You can use the code under the Regression as an example on how to use the Ridge and the Lasso methods, see the "regression slides":"https://compphysics.github.io/MachineLearning/doc/pub/Regression/html/Regression-bs.html"). + +We will thus again generate our own dataset for a function $y(x)$ where +$x \in [0,1]$ and defined by random numbers computed with the uniform +distribution. The function $y$ is a quadratic polynomial in $x$ with +added stochastic noise according to the normal distribution $\cal{N}(0,1)$. + +The following simple Python instructions define our $x$ and $y$ values (with 100 data points). +!bc pycod +x = np.random.rand(100,1) +y = 5*x*x+0.1*np.random.randn(100,1) +!ec + +o Write your own code for the Ridge method (see chapter 3.4 of Hastie *et al.*, equations (3.43) and (3.44)) and compute the parametrization for different values of $\lambda$. Compare and analyze your results with those from exercise 2. Study the dependence on $\lambda$ while also varying the strength of the noise in your expression for $y(x)$. + +o Repeat the above but using the functionality of _scikit-learn_. Compare your code with the results from _scikit-learn_. Remember to run with the same random numbers for generating $x$ and $y$. + +o Our next step is to study the variance of the parameters $\beta_1$ and $\beta_2$ (assuming that we are parametrizing our function with a second-order polynomial. We will use standard linear regression and the Ridge regression. You can now opt for either writing your own function that calculates the variance of these paramaters (recall that this is equal to the diagonal elements of the matrix $(\hat{X}^T\hat{X})+\lambda\hat{I})^{-1}$) or use the functionality of _scikit-learn_ and compute their variances. Discuss the results of these variances as functions of $\lambda$. In particular, try to link your discussion with the discussion in Hastie *et al.* and their figure 3.11. + +o Repeat the previous step but add now the Lasso method, see equation (3.53) of Hastie *et al.*. Discuss your results and compare with standard regression and the Ridge regression results. You can write your own code or use the functionality of _scikit-learn_. + +o Finally, using _scikit-learn_ or your own code, compute also the mean square error, a risk metric corresponding to the expected value of the squared (quadratic) error defined as +!bt +\[ MSE(\hat{y},\hat{\tilde{y}}) = \frac{1}{n} +\sum_{i=0}^{n-1}(y_i-\tilde{y}_i)^2, +\] +!et +and the $R^2$ score function. +If $\tilde{\hat{y}}_i$ is the predicted value of the $i-th$ sample and $y_i$ is the corresponding true value, then the score $R^2$ is defined as +!bt +\[ +R^2(\hat{y}, \tilde{\hat{y}}) = 1 - \frac{\sum_{i=0}^{n - 1} (y_i - \tilde{y}_i)^2}{\sum_{i=0}^{n - 1} (y_i - \bar{y})^2}, +\] +!et +where we have defined the mean value of $\hat{y}$ as +!bt +\[ +\bar{y} = \frac{1}{n} \sum_{i=0}^{n - 1} y_i. +\] +!et +Discuss these quantities as functions of the variable $\lambda$ in the Ridge and Lasso regression methods. + +===== Exercise 5 ===== + +Using the singular value decomposition, show that the variance of the direction vector +$\hat{z}_i=\hat{X}\hat{v}_i=\hat{u}_1d_1$ is equal to (equation (3.49) of Hastie *et al.*) +!bt +\[ +\mathrm{Var}(\hat{z}_i)=\frac{d_i^2}{N}, +\] +!et +where $d_i$ are the singular values of the matrix $\hat{X}$. In Hastie *et al*, the matrix elements of $X$ are centered. The consequence is that the mean values of for example $\hat{u}_i$ are zero. + +Give an interpretation of these results, in particular in connection with the variance of the coefficients you obtained in the previous exercise. diff --git a/doc/src/Projects/2019/Exercises/make.sh b/doc/src/Projects/2019/Exercises/make.sh new file mode 100755 index 000000000..e1fc0d6fa --- /dev/null +++ b/doc/src/Projects/2019/Exercises/make.sh @@ -0,0 +1,79 @@ +#!/bin/sh +set -x + +function system { + "$@" + if [ $? -ne 0 ]; then + echo "make.sh: unsuccessful command $@" + echo "abort!" + exit 1 + fi +} + +if [ $# -eq 0 ]; then +echo 'bash make.sh slides1|slides2' +exit 1 +fi + +name=$1 +rm -f *.tar.gz + +opt="--encoding=utf-8" +opt= + +rm -f *.aux + + + +# Plain HTML documents +html=${name} +system doconce format html $name --pygments_html_style=default --html_style=bloodish --html_links_in_new_window --html_output=$html $opt +system doconce split_html $html.html --method=space10 + +# Bootstrap style +html=${name}-bs +system doconce format html $name --html_style=bootstrap --pygments_html_style=default --html_admon=bootstrap_panel --html_output=$html $opt +system doconce split_html $html.html --method=split --pagination --nav_button=bottom + + +# Ordinary plain LaTeX document +system doconce format pdflatex $name --print_latex_style=trac --latex_admon=paragraph $opt +system doconce ptex2tex $name envir=print +# Add special packages +doconce subst "% Add user's preamble" "\g<1>\n\\usepackage{simplewick}" $name.tex +doconce replace 'section{' 'section*{' $name.tex +pdflatex -shell-escape $name +pdflatex -shell-escape $name +mv -f $name.pdf ${name}.pdf +cp $name.tex ${name}.tex + +# Publish +dest=../../../../Projects/2019 +if [ ! -d $dest/$name ]; then +mkdir $dest/$name +mkdir $dest/$name/pdf +mkdir $dest/$name/html +mkdir $dest/$name/ipynb +fi +cp ${name}*.tex $dest/$name/pdf +cp ${name}*.pdf $dest/$name/pdf +cp -r ${name}*.html ._${name}*.html $dest/$name/html + +# Figures: cannot just copy link, need to physically copy the files +if [ -d fig-${name} ]; then +if [ ! -d $dest/$name/html/fig-$name ]; then +mkdir $dest/$name/html/fig-$name +fi +cp -r fig-${name}/* $dest/$name/html/fig-$name +fi + +cp ${name}.ipynb $dest/$name/ipynb +ipynb_tarfile=ipynb-${name}-src.tar.gz +if [ ! -f ${ipynb_tarfile} ]; then +cat > README.txt <\n\\usepackage{simplewick}" $name.tex +doconce replace 'section{' 'section*{' $name.tex +pdflatex -shell-escape $name +pdflatex -shell-escape $name +mv -f $name.pdf ${name}.pdf +cp $name.tex ${name}.tex + +# Publish +dest=../../../../Projects/2018 +if [ ! -d $dest/$name ]; then +mkdir $dest/$name +mkdir $dest/$name/pdf +mkdir $dest/$name/html +mkdir $dest/$name/ipynb +fi +cp ${name}*.tex $dest/$name/pdf +cp ${name}*.pdf $dest/$name/pdf +cp -r ${name}*.html ._${name}*.html $dest/$name/html + +# Figures: cannot just copy link, need to physically copy the files +if [ -d fig-${name} ]; then +if [ ! -d $dest/$name/html/fig-$name ]; then +mkdir $dest/$name/html/fig-$name +fi +cp -r fig-${name}/* $dest/$name/html/fig-$name +fi + +cp ${name}.ipynb $dest/$name/ipynb +ipynb_tarfile=ipynb-${name}-src.tar.gz +if [ ! -f ${ipynb_tarfile} ]; then +cat > README.txt <= {} + {} = {}'.format(error[degree], bias[degree], variance[degree], bias[degree]+variance[degree])) -plt.plot(polydegree, np.log10(error), label='Error') +plt.plot(polydegree, error, label='Error') plt.plot(polydegree, bias, label='bias') plt.plot(polydegree, variance, label='Variance') plt.legend()